diff --git a/config.template b/config.template index 524a377..efe9409 100644 --- a/config.template +++ b/config.template @@ -56,8 +56,8 @@ wikipedia = True # Enable ollama LLM see more at https://ollama.com ollama = False -# Ollama model to use (defaults to gemma2:2b) -# ollamaModel = llama3.1 +# Ollama model to use (defaults to gemma3:270m) +# ollamaModel = gemma3:latest # server instance to use (defaults to local machine install) ollamaHostName = http://localhost:11434 # Produce LLM replies to messages that aren't commands? diff --git a/install.sh b/install.sh index 4c44c2f..7c49cce 100755 --- a/install.sh +++ b/install.sh @@ -250,7 +250,7 @@ if [[ $(echo "${embedded}" | grep -i "^n") ]]; then printf "\nOptionally if you want to install the multi gig LLM Ollama compnents we will execute the following commands\n" printf "\ncurl -fsSL https://ollama.com/install.sh | sh\n" - printf "ollama pull gemma2:2b\n" + printf "ollama pull gemma3:latest\n" printf "Total download is multi GB, recomend pi5/8GB or better for this\n" # ask if the user wants to install the LLM Ollama components printf "\nDo you want to install the LLM Ollama components? (y/n)" @@ -258,12 +258,12 @@ if [[ $(echo "${embedded}" | grep -i "^n") ]]; then if [[ $(echo "${ollama}" | grep -i "^y") ]]; then curl -fsSL https://ollama.com/install.sh | sh - # ask if want to install gemma2:2b - printf "\n Ollama install done now we can install the Gemma2:2b components\n" - echo "Do you want to install the Gemma2:2b components? (y/n)" + # ask if want to install gemma3:latest + printf "\n Ollama install done now we can install the gemma3:latest components\n" + echo "Do you want to install the gemma3:latest components? (y/n)" read gemma if [[ $(echo "${gemma}" | grep -i "^y") ]]; then - ollama pull gemma2:2b + ollama pull gemma3:latest fi fi diff --git a/modules/llm.py b/modules/llm.py index b0c36fb..60bb055 100644 --- a/modules/llm.py +++ b/modules/llm.py @@ -10,25 +10,16 @@ import requests import json from googlesearch import search # pip install googlesearch-python -# This is my attempt at a simple RAG implementation it will require some setup -# you will need to have the RAG data in a folder named rag in the data directory (../data/rag) -# This is lighter weight and can be used in a standalone environment, needs chromadb -# "chat with a file" is the use concept here, the file is the RAG data -# is anyone using this please let me know if you are Dec62024 -kelly -ragDEV = False - -if ragDEV: - import os - import ollama # pip install ollama - import chromadb # pip install chromadb - from ollama import Client as OllamaClient - ollamaClient = OllamaClient(host=ollamaHostName) - # LLM System Variables ollamaAPI = ollamaHostName + "/api/generate" +rawQuery = True # if True, the input is sent raw to the LLM, if False, it is processed by the meshBotAI template + openaiAPI = "https://api.openai.com/v1/completions" # not used, if you do push a enhancement! + +# Used in the meshBotAI template llmEnableHistory = True # enable last message history for the LLM model llmContext_fromGoogle = True # enable context from google search results adds to compute time but really helps with responses accuracy + googleSearchResults = 3 # number of google search results to include in the context more results = more compute time antiFloodLLM = [] llmChat_history = {} @@ -74,73 +65,6 @@ if llmEnableHistory: """ -def llm_readTextFiles(): - # read .txt files in ../data/rag - try: - text = [] - directory = "../data/rag" - for filename in os.listdir(directory): - if filename.endswith(".txt"): - filepath = os.path.join(directory, filename) - with open(filepath, 'r') as f: - text.append(f.read()) - return text - except Exception as e: - logger.debug(f"System: LLM readTextFiles: {e}") - return False - -def store_text_embedding(text): - try: - # store each document in a vector embedding database - for i, d in enumerate(text): - response = ollama.embeddings(model="mxbai-embed-large", prompt=d) - embedding = response["embedding"] - collection.add( - ids=[str(i)], - embeddings=[embedding], - documents=[d] - ) - - except Exception as e: - logger.debug(f"System: Embedding failed: {e}") - return False - -## INITALIZATION of RAG -if ragDEV: - try: - chromaHostname = "localhost:8000" - # connect to the chromaDB - chromaHost = chromaHostname.split(":")[0] - chromaPort = chromaHostname.split(":")[1] - if chromaHost == "localhost" and chromaPort == "8000": - # create a client using local python Client - chromaClient = chromadb.Client() - else: - # create a client using the remote python Client - # this isnt tested yet please test and report back - chromaClient = chromadb.Client(host=chromaHost, port=chromaPort) - - clearCollection = False - if "meshBotAI" in chromaClient.list_collections() and clearCollection: - logger.debug(f"System: LLM: Clearing RAG files from chromaDB") - chromaClient.delete_collection("meshBotAI") - - # create a new collection - collection = chromaClient.create_collection("meshBotAI") - - logger.debug(f"System: LLM: Cataloging RAG data") - store_text_embedding(llm_readTextFiles()) - - except Exception as e: - logger.debug(f"System: LLM: RAG Initalization failed: {e}") - -def query_collection(prompt): - # generate an embedding for the prompt and retrieve the most relevant doc - response = ollama.embeddings(prompt=prompt, model="mxbai-embed-large") - results = collection.query(query_embeddings=[response["embedding"]], n_results=1) - data = results['documents'][0][0] - return data - def llm_query(input, nodeID=0, location_name=None): global antiFloodLLM, llmChat_history googleResults = [] @@ -162,7 +86,7 @@ def llm_query(input, nodeID=0, location_name=None): else: antiFloodLLM.append(nodeID) - if llmContext_fromGoogle: + if llmContext_fromGoogle and not rawQuery: # grab some context from the internet using google search hits (if available) # localization details at https://pypi.org/project/googlesearch-python/ @@ -193,36 +117,27 @@ def llm_query(input, nodeID=0, location_name=None): location_name += f" at the current time of {datetime.now().strftime('%Y-%m-%d %H:%M:%S %Z')}" try: - # RAG context inclusion testing - ragContext = False - if ragDEV: - ragContext = query_collection(input) - - if ragContext: - ragContextGooogle = ragContext + '\n'.join(googleResults) - # Build the query from the template - modelPrompt = meshBotAI.format(input=input, context=ragContext, location_name=location_name, llmModel=llmModel, history=history) - # Query the model with RAG context - result = ollamaClient.generate(model=llmModel, prompt=modelPrompt) - # Condense the result to just needed - if isinstance(result, dict): - result = result.get("response") + if rawQuery: + # sanitize the input to remove tool call syntax + input = input.replace('```', '').replace('```bash', '').replace('```python', '') + modelPrompt = input else: # Build the query from the template modelPrompt = meshBotAI.format(input=input, context='\n'.join(googleResults), location_name=location_name, llmModel=llmModel, history=history) - llmQuery = {"model": llmModel, "prompt": modelPrompt, "stream": False} - # Query the model via Ollama web API - result = requests.post(ollamaAPI, data=json.dumps(llmQuery)) - # Condense the result to just needed - if result.status_code == 200: - result_json = result.json() - result = result_json.get("response", "") + + llmQuery = {"model": llmModel, "prompt": modelPrompt, "stream": False} + # Query the model via Ollama web API + result = requests.post(ollamaAPI, data=json.dumps(llmQuery)) + # Condense the result to just needed + if result.status_code == 200: + result_json = result.json() + result = result_json.get("response", "") - # deepseek-r1 has added tags to the response - if "" in result: - result = result.split("")[1] - else: - raise Exception(f"HTTP Error: {result.status_code}") + # deepseek-r1 has added tags to the response + if "" in result: + result = result.split("")[1] + else: + raise Exception(f"HTTP Error: {result.status_code}") #logger.debug(f"System: LLM Response: " + result.strip().replace('\n', ' ')) except Exception as e: diff --git a/modules/settings.py b/modules/settings.py index 50e871d..74ae700 100644 --- a/modules/settings.py +++ b/modules/settings.py @@ -219,7 +219,7 @@ try: solar_conditions_enabled = config['general'].getboolean('spaceWeather', True) wikipedia_enabled = config['general'].getboolean('wikipedia', False) llm_enabled = config['general'].getboolean('ollama', False) # https://ollama.com - llmModel = config['general'].get('ollamaModel', 'gemma2:2b') # default gemma2:2b + llmModel = config['general'].get('ollamaModel', 'gemma3:270m') # default gemma3:270m ollamaHostName = config['general'].get('ollamaHostName', 'http://localhost:11434') # default localhost llmReplyToNonCommands = config['general'].getboolean('llmReplyToNonCommands', True) dont_retry_disconnect = config['general'].getboolean('dont_retry_disconnect', False) # default False, retry on disconnect