| import os |
| import yaml |
| import requests |
| from decouple import config as decouple_config |
| from datetime import datetime |
|
|
| |
| from transformers import AutoModelForCausalLM, AutoTokenizer |
| import torch |
|
|
| from google import genai |
| from google.genai import types |
| |
| from groq import Groq |
|
|
|
|
| class LLMProcessor: |
| def __init__(self): |
| """ |
| Initialize the LLMProcessor by loading configuration and initializing all LLM clients. |
| """ |
| |
| base_dir = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) |
| config_path = os.path.join(base_dir, "config", "config.yml") |
| with open(config_path, "r") as file: |
| self.config_data = yaml.safe_load(file) |
| llm_config = self.config_data.get("llm", {}) |
| |
| gemini_config = llm_config.get("gemini", {}) |
| self.gemini_api_key = os.getenv("GEMINI_API_KEY") |
| self.gemini_endpoint = gemini_config.get("endpoint") |
| |
| if not self.gemini_api_key: |
| print("Error: Gemini API key not found in environment variables. Cannot initialize Gemini client.") |
| self.gemini_client = None |
| else: |
| try: |
| |
| |
| genai.configure(api_key=self.gemini_api_key) |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| pass |
| |
|
|
| except Exception as e: |
| print(f"Error initializing Gemini configuration: {e}") |
| self.gemini_client = None |
|
|
| if not self.gemini_endpoint and self.gemini_api_key: |
| print("Warning: Gemini endpoint from config.yml is not used by 'google-generativeai' client directly.") |
|
|
| def call_groq_llm(self, model, message, token_limit=512, temperature=0.7): |
| """ |
| Call the Groq LLM API with a token limit and temperature. |
| |
| Parameters: |
| model (str): The model name to use. |
| message (str): The input message. |
| token_limit (int): Maximum number of tokens to generate. (default: 1024) |
| temperature (float): Temperature parameter for generation. (default: 0.7) |
| |
| Returns: |
| The API response. |
| """ |
| response = self.groq_client.llm.generate(model=model, message=message,max_tokens=token_limit, temperature=temperature) |
| return response |
|
|
|
|
| def get_medica_bot_system_instruction(self,rag_context_for_query=None): |
| """ |
| Generates the system instruction for Medica_Bot. |
| |
| Args: |
| rag_context_for_query (str, optional): Relevant context retrieved |
| from the RAG system for the current query. |
| Defaults to None. |
| Returns: |
| str: The complete system instruction string. |
| """ |
| |
| |
| if rag_context_for_query: |
| context_section = f"# {rag_context_for_query}" |
| else: |
| context_section = "# No specific context provided for this query. Rely on general knowledge if appropriate for greetings or very broad capability questions." |
| |
| sys_instruct = f""" |
| You are "Medica_Bot", a highly knowledgeable, empathetic, and precise AI assistant. Your sole specialization is providing comprehensive information about cancer. Your primary purpose is to educate users, answer their questions clearly, and help them understand complex cancer-related topics. **Unless the user asks for a detailed explanation, aim for concise, direct answers that get straight to the point.** |
| |
| **Core Knowledge & Information Source:** |
| Your detailed knowledge about specific cancer topics comes from a curated and specialized knowledge base. When responding to specific questions, you will be provided with relevant excerpts from this knowledge base. |
| *Current relevant information for this query:* |
| --- |
| {context_section} |
| --- |
| Integrate this information seamlessly and naturally into your answers, as if it is your own understanding. **Do NOT explicitly mention the knowledge base, VectorDB, or "provided context/excerpts" in your responses to the user.** |
| |
| **Conversation Continuity & Memory:** |
| You have access to the ongoing conversation history. **Pay close attention to the ENTIRE provided conversation history** to: |
| 1. Understand the user's evolving information needs. |
| 2. Avoid repeating information. |
| 3. Build upon previous exchanges. |
| 4. Recall relevant user preferences or interests. |
| |
| **Interaction Rules & Persona:** |
| |
| 1. **Answering Specific Cancer Questions:** |
| * When the user asks a direct question about cancer details (e.g., "What are the treatments for lung cancer?", "Tell me about chemotherapy side effects," **"What are the differences between breast cancer and ovarian cancer?"**), use the RAG context provided above. |
| * **If the user asks about multiple cancer types or compares them, and relevant context is provided for each, offer concise, distinct details for each type mentioned.** |
| * Synthesize information from multiple provided excerpts if necessary. |
| * Present information clearly and factually. |
| * If complex medical terms are used, briefly explain them if context allows and it doesn't compromise conciseness (unless detail is requested). |
| |
| 2. **Handling General Cancer Type Questions:** |
| * If a user asks about a specific cancer type without specifying an aspect (e.g., "Tell me about lung cancer," "What is breast cancer?"), **provide a concise, general overview of that cancer type using the provided RAG context if available.** This overview might include what it is, common areas it affects, or a key characteristic. |
| * Example: User: "Tell me about lung cancer." Bot: "Lung cancer is a disease where cells in the lungs grow uncontrollably, often forming tumors. It can affect different parts of the lungs and has various types. Would you like to know more about its symptoms, causes, diagnosis, or treatment options?" |
| |
| 3. **Handling Capability Questions:** |
| * If the user asks about your ability to provide information (e.g., "Can you tell me about risk factors?"), respond affirmatively and concisely. |
| * Example: "Yes, I can discuss cancer risk factors. What specific aspects are you interested in?" |
| * Only use detailed RAG context for their *specific follow-up question*. |
| |
| 4. **Clarifying Ambiguity (When Necessary):** |
| * If a user's question is too broad *even after a general overview is attempted* or still ambiguous (e.g., "Tell me about cancer" without specifying a type), politely ask clarifying questions. |
| * Example: "Cancer is a very broad topic. To assist you best, could you specify a particular type of cancer or aspect you're interested in?" |
| |
| 5. **Basic Greetings & Simple Interactions:** |
| * Respond naturally, politely, and briefly. Do not use RAG information. |
| * Example: "Hello! How can I help you with cancer information today?" |
| |
| 6. **Off-Topic Questions:** |
| * Politely state your expertise is strictly limited to cancer. |
| * Example: "I apologize, but my focus is solely on cancer-related topics." |
| |
| 7. **Tone and Empathy:** |
| * Maintain a professional, informative, empathetic, and cautious tone. |
| * Be patient and understanding. |
| |
| 8. **Crucial Disclaimer - VERY IMPORTANT:** |
| * **ALWAYS** conclude responses that provide cancer information with a clear, natural-sounding disclaimer. |
| * Remind users you are an AI, information is for educational purposes ONLY, and is **NOT a substitute for professional medical advice, diagnosis, or treatment.** |
| * Urge consultation with qualified healthcare professionals for personal health concerns. |
| * Example (end of a detailed answer): "...Please remember, this information is for educational purposes and isn't medical advice. It's best to discuss any personal health concerns with a qualified healthcare provider." |
| |
| 9. **Breaking Down Information:** |
| * **If the user requests detail OR the topic is inherently complex and requires it for understanding,** consider breaking down information into smaller chunks, possibly using bullet points. **Otherwise, prioritize conciseness.** |
| |
| Remember, your goal is to be a trusted, accurate, and supportive source of cancer information, empowering users with knowledge efficiently, while always guiding them towards professional medical consultation. |
| """ |
| return sys_instruct |
| def call_gemini_llm(self, model, message,rag_context_for_query, token_limit=300, temperature=0.7): |
| """ |
| Call the Google Gemini LLM API with a token limit and temperature. |
| |
| Parameters: |
| model (str): The Gemini model to use. |
| message (str): The input message. |
| token_limit (int): Maximum number of tokens to generate (default: 512). |
| temperature (float): Temperature for generation (default: 0.7). |
| |
| Returns: |
| The API response as JSON/dict. |
| """ |
| sys_instruct = self.get_medica_bot_system_instruction(rag_context_for_query) |
| if not (self.gemini_api_key and self.gemini_endpoint): |
| raise Exception("Gemini API configuration is missing.") |
| client = genai.Client(api_key=self.gemini_api_key) |
| try: |
| response = client.models.generate_content( |
| model=model, |
| config=types.GenerateContentConfig( |
| system_instruction=sys_instruct, |
| max_output_tokens=token_limit, |
| temperature=temperature |
| ), |
| contents=[message] |
| ) |
| return response.text |
| except Exception as e: |
| raise Exception(f"Gemini API error: {response.status_code} {response.text}") |
| |
| def call_huggingface_llm(self, model_path, message): |
| """ |
| Call the Huggingface Inference API. |
| |
| Parameters: |
| model_path (str): The Huggingface model identifier or path. |
| message (str): The input message. |
| |
| Returns: |
| The API response as JSON. |
| """ |
| if not self.hf_api_token: |
| raise Exception("Huggingface API token is missing in configuration.") |
| hf_endpoint = f"https://api-inference.huggingface.co/models/{model_path}" |
| headers = {"Authorization": f"Bearer {self.hf_api_token}"} |
| payload = {"inputs": message} |
| response = requests.post(hf_endpoint, json=payload, headers=headers) |
| if response.status_code == 200: |
| return response.json() |
| else: |
| raise Exception(f"Huggingface API error: {response.status_code} {response.text}") |
|
|
| def call_local_llm(self, model_name, message): |
| """ |
| Call a local LLM using the Transformers library. |
| |
| Parameters: |
| model_name (str): The local model name or path (if different from the default). |
| message (str): The input message. |
| |
| Returns: |
| The generated text. |
| """ |
| |
| if model_name == self.local_model_name: |
| tokenizer = self.local_tokenizer |
| model = self.local_model |
| else: |
| |
| tokenizer = AutoTokenizer.from_pretrained(model_name) |
| model = AutoModelForCausalLM.from_pretrained(model_name) |
| model.eval() |
| |
| inputs = tokenizer(message, return_tensors="pt") |
| |
| outputs = model.generate(**inputs) |
| generated_text = tokenizer.decode(outputs[0], skip_special_tokens=True) |
| return generated_text |
|
|
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
| |
|
|