diff --git a/config.template b/config.template index e94c846..0e8a13c 100644 --- a/config.template +++ b/config.template @@ -57,6 +57,13 @@ spaceWeather = True # enable or disable the wikipedia search module wikipedia = True +# Use local Kiwix server instead of online Wikipedia +# Set to False to use online Wikipedia, or provide Kiwix server URL +useKiwixServer = False +# Kiwix server URL (e.g., http://127.0.0.1:8080) +kiwixURL = http://127.0.0.1:8080 +# Kiwix library name (e.g., wikipedia_en_100_nopic_2024-06) +kiwixLibraryName = wikipedia_en_100_nopic_2024-06 # Enable ollama LLM see more at https://ollama.com ollama = False diff --git a/modules/settings.py b/modules/settings.py index 82de6cf..d6c5346 100644 --- a/modules/settings.py +++ b/modules/settings.py @@ -226,6 +226,9 @@ try: bee_enabled = config['general'].getboolean('bee', False) # 🐝 off by default undocumented solar_conditions_enabled = config['general'].getboolean('spaceWeather', True) wikipedia_enabled = config['general'].getboolean('wikipedia', False) + use_kiwix_server = config['general'].getboolean('useKiwixServer', False) + kiwix_url = config['general'].get('kiwixURL', 'http://127.0.0.1:8080') + kiwix_library_name = config['general'].get('kiwixLibraryName', 'wikipedia_en_100_nopic_2024-06') llm_enabled = config['general'].getboolean('ollama', False) # https://ollama.com ollamaHostName = config['general'].get('ollamaHostName', 'http://localhost:11434') # default localhost llmModel = config['general'].get('ollamaModel', 'gemma3:270m') # default gemma3:270m diff --git a/modules/system.py b/modules/system.py index 71c0b46..67d60ab 100644 --- a/modules/system.py +++ b/modules/system.py @@ -209,6 +209,13 @@ if wikipedia_enabled: import wikipedia # pip install wikipedia trap_list = trap_list + ("wiki:", "wiki?",) help_message = help_message + ", wiki:" + + # Kiwix support for local wiki + if use_kiwix_server: + import requests + from bs4 import BeautifulSoup + from urllib.parse import quote + from bs4.element import Comment # LLM Configuration if llm_enabled: @@ -753,7 +760,86 @@ def send_message(message, ch, nodeid=0, nodeInt=1, bypassChuncking=False): interface.sendText(text=message, channelIndex=ch, destinationId=nodeid) return True +# Kiwix helper functions (only loaded if use_kiwix_server is True) +if wikipedia_enabled and use_kiwix_server: + def tag_visible(element): + """Filter visible text from HTML elements for Kiwix""" + if element.parent.name in ['style', 'script', 'head', 'title', 'meta', '[document]']: + return False + if isinstance(element, Comment): + return False + return True + + def text_from_html(body): + """Extract visible text from HTML content""" + soup = BeautifulSoup(body, 'html.parser') + texts = soup.findAll(string=True) + visible_texts = filter(tag_visible, texts) + return " ".join(t.strip() for t in visible_texts if t.strip()) + + def get_kiwix_summary(search_term): + """Query local Kiwix server for Wikipedia article""" + try: + search_encoded = quote(search_term) + # Try direct article access first + wiki_article = search_encoded.capitalize().replace("%20", "_") + exact_url = f"{kiwix_url}/raw/{kiwix_library_name}/content/A/{wiki_article}" + + response = requests.get(exact_url, timeout=urlTimeoutSeconds) + if response.status_code == 200: + # Extract and clean text + text = text_from_html(response.text) + # Remove common Wikipedia metadata prefixes + text = text.split("Jump to navigation", 1)[-1] + text = text.split("Jump to search", 1)[-1] + # Truncate to reasonable length (first few sentences) + sentences = text.split('. ') + summary = '. '.join(sentences[:wiki_return_limit]) + if summary and not summary.endswith('.'): + summary += '.' + return summary.strip()[:500] # Hard limit at 500 chars + + # If direct access fails, try search + search_url = f"{kiwix_url}/search?content={kiwix_library_name}&pattern={search_encoded}" + response = requests.get(search_url, timeout=urlTimeoutSeconds) + + if response.status_code == 200 and "No results were found" not in response.text: + soup = BeautifulSoup(response.text, 'html.parser') + links = [a['href'] for a in soup.find_all('a', href=True) if "start=" not in a['href']] + + for link in links[:3]: # Check first 3 results + article_name = link.split("/")[-1] + if not article_name or article_name[0].islower(): + continue + + article_url = f"{kiwix_url}{link}" + article_response = requests.get(article_url, timeout=urlTimeoutSeconds) + if article_response.status_code == 200: + text = text_from_html(article_response.text) + text = text.split("Jump to navigation", 1)[-1] + text = text.split("Jump to search", 1)[-1] + sentences = text.split('. ') + summary = '. '.join(sentences[:wiki_return_limit]) + if summary and not summary.endswith('.'): + summary += '.' + return summary.strip()[:500] + + logger.warning(f"System: No Kiwix Results for:{search_term}") + return ERROR_FETCHING_DATA + + except requests.RequestException as e: + logger.warning(f"System: Kiwix connection error: {e}") + return "Unable to connect to local wiki server" + except Exception as e: + logger.warning(f"System: Error with Kiwix for:{search_term} {e}") + return ERROR_FETCHING_DATA + def get_wikipedia_summary(search_term): + # Use Kiwix if configured + if use_kiwix_server: + return get_kiwix_summary(search_term) + + # Otherwise use online Wikipedia wikipedia_search = wikipedia.search(search_term, results=3) wikipedia_suggest = wikipedia.suggest(search_term) #wikipedia_aroundme = wikipedia.geosearch(location[0], location[1], results=3)