diff --git a/README.md b/README.md index 9143d63..b12e658 100644 --- a/README.md +++ b/README.md @@ -211,6 +211,7 @@ defaultChannel = 0 ignoreDefaultChannel = False # ignoreDefaultChannel, the bot will ignore the default channel set above ignoreChannels = # ignoreChannels is a comma separated list of channels to ignore, e.g. 4,5 cmdBang = False # require ! to be the first character in a command +explicitCmd = True # require explicit command, the message will only be processed if it starts with a command word disable to get more activity ``` ### Location Settings @@ -347,12 +348,12 @@ repeater_channels = [2, 3] ``` ### Ollama (LLM/AI) Settings -For Ollama to work, the command line `ollama run 'model'` needs to work properly. Ensure you have enough RAM and your GPU is working as expected. The default model for this project is set to `gemma2:2b`. Ollama can be remote [Ollama Server](https://github.com/ollama/ollama/blob/main/docs/faq.md#how-do-i-configure-ollama-server) works on a pi58GB with 40 second or less response time. +For Ollama to work, the command line `ollama run 'model'` needs to work properly. Ensure you have enough RAM and your GPU is working as expected. The default model for this project is set to `gemma3:270m`. Ollama can be remote [Ollama Server](https://github.com/ollama/ollama/blob/main/docs/faq.md#how-do-i-configure-ollama-server) works on a pi58GB with 40 second or less response time. ```ini # Enable ollama LLM see more at https://ollama.com ollama = True # Ollama model to use (defaults to gemma2:2b) -ollamaModel = gemma2 #ollamaModel = llama3.1 +ollamaModel = gemma3:latest # Ollama model to use (defaults to gemma3:270m) ollamaHostName = http://localhost:11434 # server instance to use (defaults to local machine install) ``` @@ -360,6 +361,9 @@ Also see `llm.py` for changing the defaults of: ```ini # LLM System Variables +rawQuery = True # if True, the input is sent raw to the LLM if False, it is processed by the meshBotAI template + +# Used in the meshBotAI template (legacy) llmEnableHistory = True # enable history for the LLM model to use in responses adds to compute time llmContext_fromGoogle = True # enable context from google search results helps with responses accuracy googleSearchResults = 3 # number of google search results to include in the context more results = more compute time diff --git a/config.template b/config.template index 524a377..6e106bf 100644 --- a/config.template +++ b/config.template @@ -36,6 +36,8 @@ ignoreDefaultChannel = False ignoreChannels = # require ! to be the first character in a command cmdBang = False +# require explicit command, the message will only be processed if it starts with a command word +explicitCmd = True # motd is reset to this value on boot motd = Thanks for using MeshBOT! Have a good day! @@ -56,13 +58,15 @@ wikipedia = True # Enable ollama LLM see more at https://ollama.com ollama = False -# Ollama model to use (defaults to gemma2:2b) -# ollamaModel = llama3.1 +# Ollama model to use (defaults to gemma3:270m) +# ollamaModel = gemma3:latest # server instance to use (defaults to local machine install) ollamaHostName = http://localhost:11434 # Produce LLM replies to messages that aren't commands? # If False, the LLM only replies to the "ask:" and "askai" commands. llmReplyToNonCommands = True +# if True, the input is sent raw to the LLM, if False uses legacy template query +rawLLMQuery = True # StoreForward Enabled and Limits StoreForward = True diff --git a/install.sh b/install.sh index 4c44c2f..7c49cce 100755 --- a/install.sh +++ b/install.sh @@ -250,7 +250,7 @@ if [[ $(echo "${embedded}" | grep -i "^n") ]]; then printf "\nOptionally if you want to install the multi gig LLM Ollama compnents we will execute the following commands\n" printf "\ncurl -fsSL https://ollama.com/install.sh | sh\n" - printf "ollama pull gemma2:2b\n" + printf "ollama pull gemma3:latest\n" printf "Total download is multi GB, recomend pi5/8GB or better for this\n" # ask if the user wants to install the LLM Ollama components printf "\nDo you want to install the LLM Ollama components? (y/n)" @@ -258,12 +258,12 @@ if [[ $(echo "${embedded}" | grep -i "^n") ]]; then if [[ $(echo "${ollama}" | grep -i "^y") ]]; then curl -fsSL https://ollama.com/install.sh | sh - # ask if want to install gemma2:2b - printf "\n Ollama install done now we can install the Gemma2:2b components\n" - echo "Do you want to install the Gemma2:2b components? (y/n)" + # ask if want to install gemma3:latest + printf "\n Ollama install done now we can install the gemma3:latest components\n" + echo "Do you want to install the gemma3:latest components? (y/n)" read gemma if [[ $(echo "${gemma}" | grep -i "^y") ]]; then - ollama pull gemma2:2b + ollama pull gemma3:latest fi fi diff --git a/mesh_bot.py b/mesh_bot.py index 1adeac1..a76db89 100755 --- a/mesh_bot.py +++ b/mesh_bot.py @@ -1162,6 +1162,8 @@ def onReceive(packet, interface): if 'decoded' in packet and packet['decoded']['portnum'] == 'TEXT_MESSAGE_APP': message_bytes = packet['decoded']['payload'] message_string = message_bytes.decode('utf-8') + via_mqtt = packet['decoded'].get('viaMqtt', False) + rx_time = packet['decoded'].get('rxTime', time.time()) # check if the packet is from us if message_from_id in [myNodeNum1, myNodeNum2, myNodeNum3, myNodeNum4, myNodeNum5, myNodeNum6, myNodeNum7, myNodeNum8, myNodeNum9]: @@ -1201,13 +1203,15 @@ def onReceive(packet, interface): if enableHopLogs: logger.debug(f"System: Packet HopDebugger: hop_away:{hop_away} hop_limit:{hop_limit} hop_start:{hop_start}") - if hop_away == 0 and hop_limit == 0 and hop_start == 0: - logger.debug(f"System: Packet HopDebugger: No hop count found in PACKET {packet} END PACKET") + + if hop_away == 0 and hop_limit == 0 and hop_start == 0: + hop = "Last Hop" + hop_count = 0 if hop_start == hop_limit: hop = "Direct" hop_count = 0 - elif hop_start == 0 and hop_limit > 0: + elif hop_start == 0 and hop_limit > 0 or via_mqtt: hop = "MQTT" hop_count = 0 else: diff --git a/modules/llm.py b/modules/llm.py index b0c36fb..d18e5ac 100644 --- a/modules/llm.py +++ b/modules/llm.py @@ -8,32 +8,34 @@ from modules.log import * # https://github.com/ollama/ollama/blob/main/docs/faq.md#how-do-i-configure-ollama-server import requests import json -from googlesearch import search # pip install googlesearch-python -# This is my attempt at a simple RAG implementation it will require some setup -# you will need to have the RAG data in a folder named rag in the data directory (../data/rag) -# This is lighter weight and can be used in a standalone environment, needs chromadb -# "chat with a file" is the use concept here, the file is the RAG data -# is anyone using this please let me know if you are Dec62024 -kelly -ragDEV = False - -if ragDEV: - import os - import ollama # pip install ollama - import chromadb # pip install chromadb - from ollama import Client as OllamaClient - ollamaClient = OllamaClient(host=ollamaHostName) +if not rawLLMQuery: + # this may be removed in the future + from googlesearch import search # pip install googlesearch-python # LLM System Variables ollamaAPI = ollamaHostName + "/api/generate" +tokens = 450 # max charcters for the LLM response, this is the max length of the response also in prompts +requestTruncation = True # if True, the LLM "will" truncate the response + openaiAPI = "https://api.openai.com/v1/completions" # not used, if you do push a enhancement! + +# Used in the meshBotAI template llmEnableHistory = True # enable last message history for the LLM model llmContext_fromGoogle = True # enable context from google search results adds to compute time but really helps with responses accuracy + googleSearchResults = 3 # number of google search results to include in the context more results = more compute time antiFloodLLM = [] llmChat_history = {} trap_list_llm = ("ask:", "askai") +meshbotAIinit = """ + keep responses as short as possible. chatbot assistant no followuyp questions, no asking for clarification. + You must respond in plain text standard ASCII characters or emojis. + """ + +truncatePrompt = f"truncate this as short as possible:\n" + meshBotAI = """ FROM {llmModel} SYSTEM @@ -74,76 +76,16 @@ if llmEnableHistory: """ -def llm_readTextFiles(): - # read .txt files in ../data/rag - try: - text = [] - directory = "../data/rag" - for filename in os.listdir(directory): - if filename.endswith(".txt"): - filepath = os.path.join(directory, filename) - with open(filepath, 'r') as f: - text.append(f.read()) - return text - except Exception as e: - logger.debug(f"System: LLM readTextFiles: {e}") - return False - -def store_text_embedding(text): - try: - # store each document in a vector embedding database - for i, d in enumerate(text): - response = ollama.embeddings(model="mxbai-embed-large", prompt=d) - embedding = response["embedding"] - collection.add( - ids=[str(i)], - embeddings=[embedding], - documents=[d] - ) - - except Exception as e: - logger.debug(f"System: Embedding failed: {e}") - return False - -## INITALIZATION of RAG -if ragDEV: - try: - chromaHostname = "localhost:8000" - # connect to the chromaDB - chromaHost = chromaHostname.split(":")[0] - chromaPort = chromaHostname.split(":")[1] - if chromaHost == "localhost" and chromaPort == "8000": - # create a client using local python Client - chromaClient = chromadb.Client() - else: - # create a client using the remote python Client - # this isnt tested yet please test and report back - chromaClient = chromadb.Client(host=chromaHost, port=chromaPort) - - clearCollection = False - if "meshBotAI" in chromaClient.list_collections() and clearCollection: - logger.debug(f"System: LLM: Clearing RAG files from chromaDB") - chromaClient.delete_collection("meshBotAI") - - # create a new collection - collection = chromaClient.create_collection("meshBotAI") - - logger.debug(f"System: LLM: Cataloging RAG data") - store_text_embedding(llm_readTextFiles()) - - except Exception as e: - logger.debug(f"System: LLM: RAG Initalization failed: {e}") - -def query_collection(prompt): - # generate an embedding for the prompt and retrieve the most relevant doc - response = ollama.embeddings(prompt=prompt, model="mxbai-embed-large") - results = collection.query(query_embeddings=[response["embedding"]], n_results=1) - data = results['documents'][0][0] - return data - def llm_query(input, nodeID=0, location_name=None): global antiFloodLLM, llmChat_history googleResults = [] + + # if this is the first initialization of the LLM the query of " " should bring meshbotAIinit OTA shouldnt reach this? + # This is for LLM like gemma and others now? + if input == " " and rawLLMQuery: + logger.warning("System: These LLM models lack a traditional system prompt, they can be verbose and not very helpful be advised.") + input = meshbotAIinit + if not location_name: location_name = "no location provided " @@ -162,7 +104,7 @@ def llm_query(input, nodeID=0, location_name=None): else: antiFloodLLM.append(nodeID) - if llmContext_fromGoogle: + if llmContext_fromGoogle and not rawLLMQuery: # grab some context from the internet using google search hits (if available) # localization details at https://pypi.org/project/googlesearch-python/ @@ -193,36 +135,29 @@ def llm_query(input, nodeID=0, location_name=None): location_name += f" at the current time of {datetime.now().strftime('%Y-%m-%d %H:%M:%S %Z')}" try: - # RAG context inclusion testing - ragContext = False - if ragDEV: - ragContext = query_collection(input) - - if ragContext: - ragContextGooogle = ragContext + '\n'.join(googleResults) - # Build the query from the template - modelPrompt = meshBotAI.format(input=input, context=ragContext, location_name=location_name, llmModel=llmModel, history=history) - # Query the model with RAG context - result = ollamaClient.generate(model=llmModel, prompt=modelPrompt) - # Condense the result to just needed - if isinstance(result, dict): - result = result.get("response") + if rawLLMQuery: + # sanitize the input to remove tool call syntax + if '```' in input: + logger.warning("System: LLM Query: Code markdown detected, removing for raw query") + input = input.replace('```bash', '').replace('```python', '').replace('```', '') + modelPrompt = input else: # Build the query from the template modelPrompt = meshBotAI.format(input=input, context='\n'.join(googleResults), location_name=location_name, llmModel=llmModel, history=history) - llmQuery = {"model": llmModel, "prompt": modelPrompt, "stream": False} - # Query the model via Ollama web API - result = requests.post(ollamaAPI, data=json.dumps(llmQuery)) - # Condense the result to just needed - if result.status_code == 200: - result_json = result.json() - result = result_json.get("response", "") + + llmQuery = {"model": llmModel, "prompt": modelPrompt, "stream": False, "max_tokens": tokens} + # Query the model via Ollama web API + result = requests.post(ollamaAPI, data=json.dumps(llmQuery)) + # Condense the result to just needed + if result.status_code == 200: + result_json = result.json() + result = result_json.get("response", "") - # deepseek-r1 has added tags to the response - if "" in result: - result = result.split("")[1] - else: - raise Exception(f"HTTP Error: {result.status_code}") + # deepseek-r1 has added tags to the response + if "" in result: + result = result.split("")[1] + else: + raise Exception(f"HTTP Error: {result.status_code}") #logger.debug(f"System: LLM Response: " + result.strip().replace('\n', ' ')) except Exception as e: @@ -231,6 +166,23 @@ def llm_query(input, nodeID=0, location_name=None): # cleanup for message output response = result.strip().replace('\n', ' ') + + if rawLLMQuery and requestTruncation and len(response) > 450: + #retryy loop to truncate the response + logger.warning(f"System: LLM Query: Response exceeded {tokens} characters, requesting truncation") + truncateQuery = {"model": llmModel, "prompt": truncatePrompt + response, "stream": False, "max_tokens": tokens} + truncateResult = requests.post(ollamaAPI, data=json.dumps(truncateQuery)) + if truncateResult.status_code == 200: + truncate_json = truncateResult.json() + result = truncate_json.get("response", "") + + else: + #use the original result if truncation fails + logger.warning("System: LLM Query: Truncation failed, using original response") + + # cleanup for message output + response = result.strip().replace('\n', ' ') + # done with the query, remove the user from the anti flood list antiFloodLLM.remove(nodeID) diff --git a/modules/settings.py b/modules/settings.py index 50e871d..f31007d 100644 --- a/modules/settings.py +++ b/modules/settings.py @@ -197,6 +197,7 @@ try: ignoreChannels = config['general'].get('ignoreChannels', '').split(',') # ignore these channels ignoreDefaultChannel = config['general'].getboolean('ignoreDefaultChannel', False) cmdBang = config['general'].getboolean('cmdBang', False) # default off + explicitCmd = config['general'].getboolean('explicitCmd', True) # default on zuluTime = config['general'].getboolean('zuluTime', False) # aka 24 hour time log_messages_to_file = config['general'].getboolean('LogMessagesToFile', False) # default off log_backup_count = config['general'].getint('LogBackupCount', 32) # default 32 days @@ -219,8 +220,9 @@ try: solar_conditions_enabled = config['general'].getboolean('spaceWeather', True) wikipedia_enabled = config['general'].getboolean('wikipedia', False) llm_enabled = config['general'].getboolean('ollama', False) # https://ollama.com - llmModel = config['general'].get('ollamaModel', 'gemma2:2b') # default gemma2:2b ollamaHostName = config['general'].get('ollamaHostName', 'http://localhost:11434') # default localhost + llmModel = config['general'].get('ollamaModel', 'gemma3:270m') # default gemma3:270m + rawLLMQuery = config['general'].getboolean('rawLLMQuery', True) #default True llmReplyToNonCommands = config['general'].getboolean('llmReplyToNonCommands', True) dont_retry_disconnect = config['general'].getboolean('dont_retry_disconnect', False) # default False, retry on disconnect # emergency response @@ -361,7 +363,7 @@ try: splitDelay = config['messagingSettings'].getfloat('splitDelay', 0) # default 0 MESSAGE_CHUNK_SIZE = config['messagingSettings'].getint('MESSAGE_CHUNK_SIZE', 160) # default 160 wantAck = config['messagingSettings'].getboolean('wantAck', False) # default False - maxBuffer = config['messagingSettings'].getint('maxBuffer', 220) # default 220 + maxBuffer = config['messagingSettings'].getint('maxBuffer', 200) # default 200 enableHopLogs = config['messagingSettings'].getboolean('enableHopLogs', False) # default False except KeyError as e: diff --git a/modules/system.py b/modules/system.py index ab6f81e..5b1ad4c 100644 --- a/modules/system.py +++ b/modules/system.py @@ -268,6 +268,7 @@ if ble_count > 1: logger.debug(f"System: Initializing Interfaces") interface1 = interface2 = interface3 = interface4 = interface5 = interface6 = interface7 = interface8 = interface9 = None retry_int1 = retry_int2 = retry_int3 = retry_int4 = retry_int5 = retry_int6 = retry_int7 = retry_int8 = retry_int9 = False +myNodeNum1 = myNodeNum2 = myNodeNum3 = myNodeNum4 = myNodeNum5 = myNodeNum6 = myNodeNum7 = myNodeNum8 = myNodeNum9 = 777 max_retry_count1 = max_retry_count2 = max_retry_count3 = max_retry_count4 = max_retry_count5 = max_retry_count6 = max_retry_count7 = max_retry_count8 = max_retry_count9 = interface_retry_count for i in range(1, 10): interface_type = globals().get(f'interface{i}_type') @@ -686,11 +687,24 @@ def messageTrap(msg): message_list=msg.split(" ") for m in message_list: for t in trap_list: - # if word in message is in the trap list, return True - if t.lower() == m.lower(): - return True - if cmdBang and m.startswith("!"): - return True + if not explicitCmd: + # if word in message is in the trap list, return True + if t.lower() == m.lower(): + if cmdBang: + if m.startswith('!'): + return True + else: + continue + return True + else: + # if the index 0 of the message is a word in the trap list, return True + if t.lower() == m.lower() and message_list.index(m) == 0: + if cmdBang: + if m.startswith('!'): + return True + else: + continue + return True # if no trap words found, run a search for near misses like ping? or cmd? for m in message_list: for t in range(len(trap_list)): diff --git a/pong_bot.py b/pong_bot.py index a703a7f..4d54829 100755 --- a/pong_bot.py +++ b/pong_bot.py @@ -254,6 +254,7 @@ def onReceive(packet, interface): if 'decoded' in packet and packet['decoded']['portnum'] == 'TEXT_MESSAGE_APP': message_bytes = packet['decoded']['payload'] message_string = message_bytes.decode('utf-8') + via_mqtt = packet['decoded'].get('viaMqtt', False) # check if the packet is from us if message_from_id == myNodeNum1 or message_from_id == myNodeNum2: @@ -283,10 +284,17 @@ def onReceive(packet, interface): else: hop_start = 0 + if enableHopLogs: + logger.debug(f"System: Packet HopDebugger: hop_away:{hop_away} hop_limit:{hop_limit} hop_start:{hop_start}") + + if hop_away == 0 and hop_limit == 0 and hop_start == 0: + hop = "Last Hop" + hop_count = 0 + if hop_start == hop_limit: hop = "Direct" hop_count = 0 - elif hop_start == 0 and hop_limit > 0: + elif hop_start == 0 and hop_limit > 0 or via_mqtt: hop = "MQTT" hop_count = 0 else: