Spaces:
Build error
Build error
| import os, json, pickle, pandas as pd, numpy as np | |
| from sklearn.metrics.pairwise import cosine_similarity | |
| from langchain_openai import AzureOpenAIEmbeddings | |
| from tqdm.auto import tqdm | |
| from typing import List, Dict, Union, Set, Optional, Any | |
| import requests | |
| from Bio.Blast import NCBIWWW, NCBIXML | |
| from Bio.Seq import Seq | |
| from Bio import Entrez | |
| # Removed biomni dependency to avoid environment variable loading side effects | |
| import traceback | |
| import time | |
| from smolagents import tool, OpenAIServerModel | |
| OPENROUTER_API_KEY_STRING = "" | |
| # Use absolute path for schema database | |
| SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) | |
| AGENTS_STELLA_DIR = os.path.dirname(SCRIPT_DIR) | |
| SCHEMA_DB_PATH = os.path.join(AGENTS_STELLA_DIR, "resource", "schema_db") | |
| def parse_hpo_obo(obo_file_path: str) -> Dict[str, str]: | |
| """ | |
| Simple HPO OBO file parser to extract term IDs and names. | |
| Args: | |
| obo_file_path (str): Path to the HPO OBO file | |
| Returns: | |
| Dict[str, str]: Dictionary mapping HPO IDs to names | |
| """ | |
| hpo_dict = {} | |
| # Handle relative paths by making them absolute relative to script location | |
| if not os.path.isabs(obo_file_path): | |
| obo_file_path = os.path.join(SCRIPT_DIR, obo_file_path) | |
| try: | |
| if not os.path.exists(obo_file_path): | |
| print(f"Warning: HPO OBO file not found at {obo_file_path}") | |
| return hpo_dict | |
| with open(obo_file_path, 'r', encoding='utf-8') as f: | |
| current_term = {} | |
| in_term_block = False | |
| for line in f: | |
| line = line.strip() | |
| if line == '[Term]': | |
| in_term_block = True | |
| current_term = {} | |
| elif line == '' and in_term_block: | |
| # End of term block | |
| if 'id' in current_term and 'name' in current_term: | |
| hpo_dict[current_term['id']] = current_term['name'] | |
| in_term_block = False | |
| elif in_term_block and ':' in line: | |
| key, value = line.split(':', 1) | |
| key = key.strip() | |
| value = value.strip() | |
| if key == 'id': | |
| current_term['id'] = value | |
| elif key == 'name': | |
| current_term['name'] = value | |
| # Handle last term if file doesn't end with empty line | |
| if in_term_block and 'id' in current_term and 'name' in current_term: | |
| hpo_dict[current_term['id']] = current_term['name'] | |
| except Exception as e: | |
| print(f"Error parsing HPO OBO file: {e}") | |
| return hpo_dict | |
| gemini_model = OpenAIServerModel( | |
| model_id="google/gemini-2.5-pro", | |
| api_base="https://openrouter.ai/api/v1", | |
| api_key=OPENROUTER_API_KEY_STRING, | |
| temperature=0.1, # Lower temperature for more consistent analysis | |
| ) | |
| def _query_gemini_for_api(prompt, schema, system_template, model=None): | |
| """ | |
| Helper function to query Gemini for generating API calls based on natural language prompts. | |
| Args: | |
| prompt (str): Natural language query to process | |
| schema (dict): API schema to include in the system prompt | |
| system_template (str): Template string for the system prompt (should have {schema} placeholder) | |
| model: Gemini model instance to use (defaults to global gemini_model) | |
| Returns: | |
| dict: Dictionary with 'success', 'data' (if successful), 'error' (if failed), and optional 'raw_response' | |
| """ | |
| # Use global gemini_model if none provided | |
| model = gemini_model | |
| try: | |
| if schema is not None: | |
| # Format the system prompt with the schema | |
| schema_json = json.dumps(schema, indent=2) | |
| system_prompt = system_template.format(schema=schema_json) | |
| else: | |
| system_prompt = system_template | |
| # Combine system prompt and user prompt for Gemini | |
| full_prompt = f"{system_prompt}\n\nUser query: {prompt}" | |
| # Create messages in the correct format for OpenAIServerModel | |
| messages = [{"role": "user", "content": full_prompt}] | |
| response = model(messages) | |
| # Extract content from ChatMessage response | |
| if hasattr(response, 'content'): | |
| gemini_text = response.content.strip() | |
| else: | |
| gemini_text = str(response).strip() | |
| # Find JSON boundaries (in case Gemini adds explanations) | |
| json_start = gemini_text.find('{') | |
| json_end = gemini_text.rfind('}') + 1 | |
| if json_start >= 0 and json_end > json_start: | |
| json_text = gemini_text[json_start:json_end] | |
| result = json.loads(json_text) | |
| else: | |
| # If no JSON found, try the whole response | |
| result = json.loads(gemini_text) | |
| return { | |
| "success": True, | |
| "data": result, | |
| "raw_response": gemini_text | |
| } | |
| except (json.JSONDecodeError, KeyError, IndexError) as e: | |
| return { | |
| "success": False, | |
| "error": f"Failed to parse Gemini's response: {str(e)}", | |
| "raw_response": gemini_text if 'gemini_text' in locals() else "No content found" | |
| } | |
| except Exception as e: | |
| return { | |
| "success": False, | |
| "error": f"Error querying Gemini: {str(e)}" | |
| } | |
| # Function to map HPO terms to names | |
| def get_hpo_names(hpo_terms: List[str], ) -> List[str]: | |
| """ | |
| Retrieve the names of given HPO terms. | |
| Args: | |
| hpo_terms (List[str]): A list of HPO terms (e.g., ['HP:0001250']). | |
| Returns: | |
| List[str]: A list of corresponding HPO term names. | |
| """ | |
| hp_dict = parse_hpo_obo(os.path.join(AGENTS_STELLA_DIR, 'resource', 'hp.obo')) | |
| hpo_names = [] | |
| for term in hpo_terms: | |
| name = hp_dict.get(term, f"Unknown term: {term}") | |
| hpo_names.append(name) | |
| return hpo_names | |
| def _query_rest_api(endpoint, method="GET", params=None, headers=None, json_data=None, description=None): | |
| """ | |
| General helper function to query REST APIs with consistent error handling. | |
| Args: | |
| endpoint (str): Full URL endpoint to query | |
| method (str): HTTP method ("GET" or "POST") | |
| params (dict, optional): Query parameters to include in the URL | |
| headers (dict, optional): HTTP headers for the request | |
| json_data (dict, optional): JSON data for POST requests | |
| description (str, optional): Description of this query for error messages | |
| Returns: | |
| dict: Dictionary containing the result or error information | |
| """ | |
| # Set default headers if not provided | |
| if headers is None: | |
| headers = {"Accept": "application/json"} | |
| # Set default description if not provided | |
| if description is None: | |
| description = f"{method} request to {endpoint}" | |
| url_error = None | |
| try: | |
| # Make the API request | |
| if method.upper() == "GET": | |
| response = requests.get(endpoint, params=params, headers=headers) | |
| elif method.upper() == "POST": | |
| response = requests.post(endpoint, params=params, headers=headers, json=json_data) | |
| else: | |
| return {"error": f"Unsupported HTTP method: {method}"} | |
| url_error = str(response.text) | |
| response.raise_for_status() | |
| # Try to parse JSON response | |
| try: | |
| result = response.json() | |
| except ValueError: | |
| # Return raw text if not JSON | |
| result = {"raw_text": response.text} | |
| return { | |
| "success": True, | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "method": method, | |
| "description": description | |
| }, | |
| "result": result | |
| } | |
| except requests.exceptions.RequestException as e: | |
| error_msg = str(e) | |
| response_text = "" | |
| # Try to get more detailed error info from response | |
| if hasattr(e, 'response') and e.response: | |
| try: | |
| error_json = e.response.json() | |
| if 'messages' in error_json: | |
| error_msg = "; ".join(error_json['messages']) | |
| elif 'message' in error_json: | |
| error_msg = error_json['message'] | |
| elif 'error' in error_json: | |
| error_msg = error_json['error'] | |
| elif 'detail' in error_json: | |
| error_msg = error_json['detail'] | |
| except: | |
| response_text = e.response.text | |
| return { | |
| "success": False, | |
| "error": f"API error: {error_msg}", | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "method": method, | |
| "description": description | |
| }, | |
| "response_url_error": url_error, | |
| "response_text": response_text | |
| } | |
| except Exception as e: | |
| return { | |
| "success": False, | |
| "error": f"Error: {str(e)}", | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "method": method, | |
| "description": description | |
| } | |
| } | |
| def _query_ncbi_database( | |
| database: str, | |
| search_term: str, | |
| result_formatter = None, | |
| max_results: int = 3, | |
| ) -> Dict[str, Any]: | |
| """ | |
| Core function to query NCBI databases using Claude for query interpretation and NCBI eutils. | |
| Args: | |
| database (str): NCBI database to query (e.g., "clinvar", "gds", "geoprofiles") | |
| result_formatter (callable): Function to format results from the database | |
| api_key (str): Anthropic API key. If None, will look for ANTHROPIC_API_KEY environment variable | |
| model (str): Anthropic model to use | |
| max_results (int): Maximum number of results to return | |
| verbose (bool): Whether to return verbose results | |
| Returns: | |
| dict: Dictionary containing both the structured query and the results | |
| """ | |
| # Query NCBI API using the structured search term | |
| esearch_url = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi" | |
| esearch_params = { | |
| "db": database, | |
| "term": search_term, | |
| "retmode": "json", | |
| "retmax": 100, | |
| "usehistory": "y" # Use history server to store results | |
| } | |
| # Get IDs of matching entries | |
| search_response = _query_rest_api( | |
| endpoint=esearch_url, | |
| method="GET", | |
| params=esearch_params, | |
| description="NCBI ESearch API query" | |
| ) | |
| if not search_response["success"]: | |
| return search_response | |
| search_data = search_response["result"] | |
| # If we have results, fetch the details | |
| if "esearchresult" in search_data and int(search_data["esearchresult"]["count"]) > 0: | |
| # Extract WebEnv and query_key from the search results | |
| webenv = search_data["esearchresult"].get("webenv", "") | |
| query_key = search_data["esearchresult"].get("querykey", "") | |
| # Use WebEnv and query_key if available | |
| if webenv and query_key: | |
| # Get details using eSummary | |
| esummary_url = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi" | |
| esummary_params = { | |
| "db": database, | |
| "query_key": query_key, | |
| "WebEnv": webenv, | |
| "retmode": "json", | |
| "retmax": max_results | |
| } | |
| details_response = _query_rest_api( | |
| endpoint=esummary_url, | |
| method="GET", | |
| params=esummary_params, | |
| description="NCBI ESummary API query" | |
| ) | |
| if not details_response["success"]: | |
| return details_response | |
| results = details_response["result"] | |
| else: | |
| # Fall back to direct ID fetch | |
| id_list = search_data["esearchresult"]["idlist"][:max_results] | |
| # Get details for each ID | |
| esummary_url = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi" | |
| esummary_params = { | |
| "db": database, | |
| "id": ",".join(id_list), | |
| "retmode": "json" | |
| } | |
| details_response = _query_rest_api( | |
| endpoint=esummary_url, | |
| method="GET", | |
| params=esummary_params, | |
| description="NCBI ESummary API query" | |
| ) | |
| if not details_response["success"]: | |
| return details_response | |
| results = details_response["result"] | |
| # Format results using the provided formatter | |
| if result_formatter: | |
| formatted_results = result_formatter(results) | |
| else: | |
| formatted_results = results | |
| # Return the combined information | |
| return { | |
| "database": database, | |
| "query_interpretation": search_term, | |
| "total_results": int(search_data["esearchresult"]["count"]), | |
| "formatted_results": formatted_results | |
| } | |
| else: | |
| return { | |
| "database": database, | |
| "query_interpretation": search_term, | |
| "total_results": 0, | |
| "formatted_results": [] | |
| } | |
| def _format_query_results(result, options=None): | |
| """ | |
| A general-purpose formatter for query function results to reduce output size. | |
| Args: | |
| result (dict): The original API response dictionary | |
| options (dict, optional): Formatting options including: | |
| - max_items (int): Maximum number of items to include in lists (default: 5) | |
| - max_depth (int): Maximum depth to traverse in nested dictionaries (default: 2) | |
| - include_keys (list): Only include these top-level keys (overrides exclude_keys) | |
| - exclude_keys (list): Exclude these keys from the output | |
| - summarize_lists (bool): Whether to summarize long lists (default: True) | |
| - truncate_strings (int): Maximum length for string values (default: 100) | |
| Returns: | |
| dict: A condensed version of the input results | |
| """ | |
| def _format_value(value, depth, options): | |
| """ | |
| Recursively format a value based on its type and formatting options. | |
| Args: | |
| value: The value to format | |
| depth (int): Current recursion depth | |
| options (dict): Formatting options | |
| Returns: | |
| Formatted value | |
| """ | |
| # Base case: reached max depth | |
| if depth >= options['max_depth'] and (isinstance(value, dict) or isinstance(value, list)): | |
| if isinstance(value, dict): | |
| return { | |
| '_summary': f'Nested dictionary with {len(value)} keys', | |
| '_keys': list(value.keys())[:options['max_items']] | |
| } | |
| else: # list | |
| return _summarize_list(value, options) | |
| # Process based on type | |
| if isinstance(value, dict): | |
| return _format_dict(value, depth, options) | |
| elif isinstance(value, list): | |
| return _format_list(value, depth, options) | |
| elif isinstance(value, str) and len(value) > options['truncate_strings']: | |
| return value[:options['truncate_strings']] + "... (truncated)" | |
| else: | |
| return value | |
| def _format_dict(d, depth, options): | |
| """Format a dictionary according to options.""" | |
| result = {} | |
| # Filter keys based on include/exclude options | |
| keys_to_process = d.keys() | |
| if depth == 0 and options['include_keys']: # Only apply at top level | |
| keys_to_process = [k for k in keys_to_process if k in options['include_keys']] | |
| elif depth == 0 and options['exclude_keys']: # Only apply at top level | |
| keys_to_process = [k for k in keys_to_process if k not in options['exclude_keys']] | |
| # Process each key | |
| for key in keys_to_process: | |
| result[key] = _format_value(d[key], depth + 1, options) | |
| return result | |
| def _format_list(lst, depth, options): | |
| """Format a list according to options.""" | |
| if options['summarize_lists'] and len(lst) > options['max_items']: | |
| return _summarize_list(lst, options) | |
| result = [] | |
| for i, item in enumerate(lst): | |
| if i >= options['max_items']: | |
| remaining = len(lst) - options['max_items'] | |
| result.append(f"... {remaining} more items (omitted)") | |
| break | |
| result.append(_format_value(item, depth + 1, options)) | |
| return result | |
| def _summarize_list(lst, options): | |
| """Create a summary for a list.""" | |
| if not lst: | |
| return [] | |
| # Sample a few items | |
| sample = lst[:min(3, len(lst))] | |
| sample_formatted = [_format_value(item, options['max_depth'], options) for item in sample] | |
| # For homogeneous lists, provide type info | |
| if len(lst) > 0: | |
| item_type = type(lst[0]).__name__ | |
| homogeneous = all(isinstance(item, type(lst[0])) for item in lst) | |
| type_info = f"all {item_type}" if homogeneous else "mixed types" | |
| else: | |
| type_info = "empty" | |
| return { | |
| '_summary': f"List with {len(lst)} items ({type_info})", | |
| '_sample': sample_formatted | |
| } | |
| if options is None: | |
| options = {} | |
| # Default options | |
| default_options = { | |
| 'max_items': 5, | |
| 'max_depth': 20, | |
| 'include_keys': None, | |
| 'exclude_keys': ['raw_response', 'debug_info', 'request_details'], | |
| 'summarize_lists': True, | |
| 'truncate_strings': 100 | |
| } | |
| # Merge provided options with defaults | |
| for key, value in default_options.items(): | |
| if key not in options: | |
| options[key] = value | |
| # Filter and format the result | |
| formatted = _format_value(result, 0, options) | |
| return formatted | |
| def query_uniprot(prompt: str = None, endpoint: str = None, max_results: int = 5) -> dict: | |
| """ | |
| Query the UniProt REST API using either natural language or a direct endpoint. | |
| Args: | |
| prompt: Natural language query about proteins (e.g., "Find information about human insulin") | |
| endpoint: Full or partial UniProt API endpoint URL to query directly | |
| max_results: Maximum number of results to return | |
| Returns: | |
| Dictionary containing the query information and the UniProt API results | |
| Examples: | |
| - Natural language: query_uniprot(prompt="Find information about human insulin protein") | |
| - Direct endpoint: query_uniprot(endpoint="https://rest.uniprot.org/uniprotkb/P01308") | |
| """ | |
| # Base URL for UniProt API | |
| base_url = "https://rest.uniprot.org" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load UniProt schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "uniprot.pkl") | |
| with open(schema_path, "rb") as f: | |
| uniprot_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a protein biology expert specialized in using the UniProt REST API. | |
| Based on the user's natural language request, determine the appropriate UniProt REST API endpoint and parameters. | |
| UNIPROT REST API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including base URL, dataset, endpoint type, and parameters) | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - Base URL is "https://rest.uniprot.org" | |
| - Search in reviewed (Swiss-Prot) entries first before using non-reviewed (TrEMBL) entries | |
| - Assume organism is human unless otherwise specified. Human taxonomy ID is 9606 | |
| - Use gene_exact: for exact gene name searches | |
| - Use specific query fields like accession:, gene:, organism_id: in search queries | |
| - Use quotes for terms with spaces: organism_name:"Homo sapiens" | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=uniprot_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Use provided endpoint directly | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = f"Direct query to provided endpoint" | |
| # Use the common REST API helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| return api_result | |
| def query_alphafold( | |
| uniprot_id: str, | |
| endpoint: str = "prediction", | |
| residue_range: str = None, | |
| download: bool = False, | |
| output_dir: str = None, | |
| file_format: str = "pdb", | |
| model_version: str = "v4", | |
| model_number: int = 1, | |
| ) -> dict: | |
| """ | |
| Query the AlphaFold Database API for protein structure predictions. | |
| Args: | |
| uniprot_id: UniProt accession ID (e.g., "P12345") | |
| endpoint: Specific AlphaFold API endpoint to query: "prediction", "summary", or "annotations" | |
| residue_range: Specific residue range in format "start-end" (e.g., "1-100") | |
| download: Whether to download structure files | |
| output_dir: Directory to save downloaded files (default: current directory) | |
| file_format: Format of the structure file to download - "pdb" or "cif" | |
| model_version: AlphaFold model version - "v4" (latest) or "v3", "v2", "v1" | |
| model_number: Model number (1-5, with 1 being the highest confidence model) | |
| Returns: | |
| Dictionary containing both the query information and the AlphaFold results | |
| Examples: | |
| - Basic query: query_alphafold(uniprot_id="P53_HUMAN") | |
| - Download structure: query_alphafold(uniprot_id="P53_HUMAN", download=True, output_dir="./structures") | |
| - Get annotations: query_alphafold(uniprot_id="P53_HUMAN", endpoint="annotations") | |
| """ | |
| # Base URL for AlphaFold API | |
| base_url = "https://alphafold.ebi.ac.uk/api" | |
| # Ensure we have a UniProt ID | |
| if not uniprot_id: | |
| return {"error": "UniProt ID is required"} | |
| # Validate endpoint | |
| valid_endpoints = ["prediction", "summary", "annotations"] | |
| if endpoint not in valid_endpoints: | |
| return {"error": f"Invalid endpoint. Must be one of: {', '.join(valid_endpoints)}"} | |
| # Construct the API URL based on endpoint | |
| if endpoint == "prediction": | |
| url = f"{base_url}/prediction/{uniprot_id}" | |
| elif endpoint == "summary": | |
| url = f"{base_url}/uniprot/summary/{uniprot_id}.json" | |
| elif endpoint == "annotations": | |
| if residue_range: | |
| url = f"{base_url}/annotations/{uniprot_id}/{residue_range}" | |
| else: | |
| url = f"{base_url}/annotations/{uniprot_id}" | |
| try: | |
| # Make the API request | |
| response = requests.get(url) | |
| response.raise_for_status() | |
| # Parse the response as JSON | |
| result = response.json() | |
| # Handle download request if specified | |
| download_info = None | |
| if download: | |
| # Ensure output directory exists | |
| if not output_dir: | |
| output_dir = "." | |
| os.makedirs(output_dir, exist_ok=True) | |
| # Generate standard AlphaFold filename | |
| file_ext = file_format.lower() | |
| filename = f"AF-{uniprot_id}-F{model_number}-model_{model_version}.{file_ext}" | |
| file_path = os.path.join(output_dir, filename) | |
| # Construct download URL | |
| download_url = f"https://alphafold.ebi.ac.uk/files/{filename}" | |
| # Download the file | |
| download_response = requests.get(download_url) | |
| if download_response.status_code == 200: | |
| with open(file_path, 'wb') as f: | |
| f.write(download_response.content) | |
| download_info = { | |
| "success": True, | |
| "file_path": file_path, | |
| "url": download_url | |
| } | |
| else: | |
| download_info = { | |
| "success": False, | |
| "error": f"Failed to download file (status code: {download_response.status_code})", | |
| "url": download_url | |
| } | |
| # Return the query information and results | |
| response_data = { | |
| "query_info": { | |
| "uniprot_id": uniprot_id, | |
| "endpoint": endpoint, | |
| "residue_range": residue_range, | |
| "url": url | |
| }, | |
| "result": result | |
| } | |
| if download_info: | |
| response_data["download"] = download_info | |
| return response_data | |
| except requests.exceptions.RequestException as e: | |
| error_msg = str(e) | |
| response_text = "" | |
| # Try to get more detailed error info from response | |
| if hasattr(e, 'response') and e.response: | |
| try: | |
| error_json = e.response.json() | |
| if 'message' in error_json: | |
| error_msg = error_json['message'] | |
| except: | |
| response_text = e.response.text | |
| return { | |
| "error": f"AlphaFold API error: {error_msg}", | |
| "query_info": { | |
| "uniprot_id": uniprot_id, | |
| "endpoint": endpoint, | |
| "residue_range": residue_range, | |
| "url": url | |
| }, | |
| "response_text": response_text | |
| } | |
| except Exception as e: | |
| return { | |
| "error": f"Error: {str(e)}", | |
| "query_info": { | |
| "uniprot_id": uniprot_id, | |
| "endpoint": endpoint, | |
| "residue_range": residue_range | |
| } | |
| } | |
| def query_interpro(prompt: str = None, endpoint: str = None, max_results: int = 3) -> dict: | |
| """ | |
| Query the InterPro REST API using natural language or a direct endpoint. | |
| Args: | |
| prompt: Natural language query about protein domains or families | |
| endpoint: Direct endpoint path or full URL (e.g., "/entry/interpro/IPR023411") | |
| max_results: Maximum number of results to return per page | |
| Returns: | |
| Dictionary containing both the query information and the InterPro API results | |
| Examples: | |
| - Natural language: query_interpro("Find information about kinase domains in InterPro") | |
| - Direct endpoint: query_interpro(endpoint="/entry/interpro/IPR023411") | |
| """ | |
| # Base URL for InterPro API | |
| base_url = "https://www.ebi.ac.uk/interpro/api" | |
| # Default parameters | |
| format = "json" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load InterPro schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "interpro.pkl") | |
| with open(schema_path, "rb") as f: | |
| interpro_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a protein domain expert specialized in using the InterPro REST API. | |
| Based on the user's natural language request, determine the appropriate InterPro REST API endpoint. | |
| INTERPRO REST API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including the base URL "https://www.ebi.ac.uk/interpro/api") | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - Path components for data types: entry, protein, structure, set, taxonomy, proteome | |
| - Common sources: interpro, pfam, cdd, uniprot, pdb | |
| - Protein subtypes can be "reviewed" or "unreviewed" | |
| - For specific entries, use lowercase accessions (e.g., "ipr000001" instead of "IPR000001") | |
| - Endpoints can be hierarchical like "/entry/interpro/protein/uniprot/P04637" | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=interpro_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Extract the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| # If it's just a path, add the base URL | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = "Direct query to provided endpoint" | |
| # Add pagination parameters | |
| params = {"page": 1, "page_size": max_results} | |
| # Add format parameter if not json | |
| if format and format != "json": | |
| params["format"] = format | |
| # Make the API request | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| params=params, | |
| description=description | |
| ) | |
| return api_result | |
| def query_pdb(prompt: str = None, query: dict = None, max_results: int = 3) -> dict: | |
| """ | |
| Query the RCSB PDB database using natural language or a direct structured query. | |
| Args: | |
| prompt: Natural language query about protein structures | |
| query: Direct structured query in RCSB Search API format (overrides prompt) | |
| max_results: Maximum number of results to return | |
| Returns: | |
| Dictionary containing the structured query, search results, and identifiers | |
| Examples: | |
| - Natural language: query_pdb("Find structures of human insulin") | |
| - Direct query: query_pdb(query={"query": {"type": "terminal", "service": "full_text", | |
| "parameters": {"value": "insulin"}}, "return_type": "entry"}) | |
| """ | |
| # Default parameters | |
| return_type = "entry" | |
| search_service = "full_text" | |
| # Generate search query from natural language if prompt is provided and query is not | |
| if prompt and not query: | |
| # Load schema from pickle file | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "pdb.pkl") | |
| with open(schema_path, "rb") as f: | |
| schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a structural biology expert that creates precise RCSB PDB Search API queries based on natural language requests. | |
| SEARCH API SCHEMA: | |
| {schema} | |
| IMPORTANT GUIDELINES: | |
| 1. Choose the appropriate search_service based on the query: | |
| - Use "text" for attribute-specific searches (REQUIRES attribute, operator, and value) | |
| - Use "full_text" for general keyword searches across multiple fields | |
| - Use appropriate specialized services for sequence, structure, motif searches | |
| 2. For "text" searches, you MUST specify: | |
| - attribute: The specific field to search (use common_attributes from schema) | |
| - operator: The comparison method (exact_match, contains_words, less_or_equal, etc.) | |
| - value: The search term or value | |
| 3. For "full_text" searches, only specify: | |
| - value: The search term(s) | |
| 4. For combined searches, use "group" nodes with logical_operator ("and" or "or") | |
| 5. Always specify the appropriate return_type based on what the user is looking for | |
| Generate a well-formed Search API query JSON object. Return ONLY the JSON with no additional explanation. | |
| """ | |
| # Query Gemini to generate the search query | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return { | |
| "error": gemini_result["error"], | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| # Get the query from Gemini's response | |
| query_json = gemini_result["data"] | |
| else: | |
| # Use provided query directly | |
| query_json = query if query else { | |
| "query": { | |
| "type": "terminal", | |
| "service": search_service, | |
| "parameters": { | |
| "value": prompt | |
| } | |
| }, | |
| "return_type": return_type | |
| } | |
| # Ensure return_type is set | |
| if "return_type" not in query_json: | |
| query_json["return_type"] = return_type | |
| # Add request options for pagination, but avoid conflicts with return_all_hits | |
| if "request_options" not in query_json: | |
| query_json["request_options"] = {} | |
| # Only add pagination if return_all_hits is not set | |
| if "return_all_hits" in query_json["request_options"] and query_json["request_options"]["return_all_hits"]: | |
| # Remove return_all_hits and use pagination instead for limited results | |
| query_json["request_options"]["return_all_hits"] = False | |
| if "paginate" not in query_json["request_options"]: | |
| query_json["request_options"]["paginate"] = { | |
| "start": 0, | |
| "rows": max_results | |
| } | |
| # Use query_rest_api to execute the search | |
| search_url = "https://search.rcsb.org/rcsbsearch/v2/query" | |
| api_result = _query_rest_api( | |
| endpoint=search_url, | |
| method="POST", | |
| json_data=query_json, | |
| description="PDB Search API query" | |
| ) | |
| return api_result | |
| def query_pdb_identifiers(identifiers: List[str], return_type: str = "entry", download: bool = False, attributes: List[str] = None) -> dict: | |
| """ | |
| Retrieve detailed data and/or download files for PDB identifiers. | |
| Args: | |
| identifiers: List of PDB identifiers (from query_pdb) | |
| return_type: Type of results: "entry", "assembly", "polymer_entity", etc. | |
| download: Whether to download PDB structure files | |
| attributes: List of specific attributes to retrieve | |
| Returns: | |
| Dictionary containing the detailed data and file paths if downloaded | |
| Example: | |
| - Search and then get details: | |
| results = query_pdb("Find structures of human insulin") | |
| details = get_pdb_details(results["identifiers"], download=True) | |
| """ | |
| if not identifiers: | |
| return {"error": "No identifiers provided"} | |
| try: | |
| # Fetch detailed data using Data API | |
| detailed_results = [] | |
| for identifier in identifiers: | |
| try: | |
| # Determine the appropriate endpoint based on return_type and identifier format | |
| if return_type == "entry": | |
| data_url = f"https://data.rcsb.org/rest/v1/core/entry/{identifier}" | |
| elif return_type == "polymer_entity": | |
| entry_id, entity_id = identifier.split('_') | |
| data_url = f"https://data.rcsb.org/rest/v1/core/polymer_entity/{entry_id}/{entity_id}" | |
| elif return_type == "nonpolymer_entity": | |
| entry_id, entity_id = identifier.split('_') | |
| data_url = f"https://data.rcsb.org/rest/v1/core/nonpolymer_entity/{entry_id}/{entity_id}" | |
| elif return_type == "polymer_instance": | |
| entry_id, asym_id = identifier.split('.') | |
| data_url = f"https://data.rcsb.org/rest/v1/core/polymer_entity_instance/{entry_id}/{asym_id}" | |
| elif return_type == "assembly": | |
| entry_id, assembly_id = identifier.split('-') | |
| data_url = f"https://data.rcsb.org/rest/v1/core/assembly/{entry_id}/{assembly_id}" | |
| elif return_type == "mol_definition": | |
| data_url = f"https://data.rcsb.org/rest/v1/core/chem_comp/{identifier}" | |
| # Fetch data | |
| data_response = requests.get(data_url) | |
| data_response.raise_for_status() | |
| entity_data = data_response.json() | |
| # Filter attributes if specified | |
| if attributes: | |
| filtered_data = {} | |
| for attr in attributes: | |
| parts = attr.split('.') | |
| current = entity_data | |
| try: | |
| for part in parts[:-1]: | |
| current = current[part] | |
| filtered_data[attr] = current[parts[-1]] | |
| except (KeyError, TypeError): | |
| filtered_data[attr] = None | |
| entity_data = filtered_data | |
| detailed_results.append({ | |
| "identifier": identifier, | |
| "data": entity_data | |
| }) | |
| except Exception as e: | |
| detailed_results.append({ | |
| "identifier": identifier, | |
| "error": str(e) | |
| }) | |
| # Download structure files if requested | |
| if download: | |
| for identifier in identifiers: | |
| if '_' in identifier or '.' in identifier or '-' in identifier: | |
| # For non-entry identifiers, extract the PDB ID | |
| if '_' in identifier: | |
| pdb_id = identifier.split('_')[0] | |
| elif '.' in identifier: | |
| pdb_id = identifier.split('.')[0] | |
| elif '-' in identifier: | |
| pdb_id = identifier.split('-')[0] | |
| else: | |
| pdb_id = identifier | |
| try: | |
| # Download PDB file | |
| pdb_url = f"https://files.rcsb.org/download/{pdb_id}.pdb" | |
| pdb_response = requests.get(pdb_url) | |
| if pdb_response.status_code == 200: | |
| # Create data directory if it doesn't exist | |
| data_dir = os.path.join(os.path.dirname(__file__), "data", "pdb") | |
| os.makedirs(data_dir, exist_ok=True) | |
| # Save PDB file | |
| pdb_file_path = os.path.join(data_dir, f"{pdb_id}.pdb") | |
| with open(pdb_file_path, 'wb') as pdb_file: | |
| pdb_file.write(pdb_response.content) | |
| # Add download information to results | |
| for result in detailed_results: | |
| if result["identifier"] == identifier or result["identifier"].startswith(pdb_id): | |
| result["pdb_file_path"] = pdb_file_path | |
| except Exception as e: | |
| for result in detailed_results: | |
| if result["identifier"] == identifier or result["identifier"].startswith(pdb_id): | |
| result["download_error"] = str(e) | |
| return { | |
| "detailed_results": detailed_results | |
| } | |
| except Exception as e: | |
| return { | |
| "error": f"Error retrieving PDB details: {str(e)}" | |
| } | |
| def query_kegg(prompt: str, endpoint: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Take a natural language prompt and convert it to a structured KEGG API query. | |
| Args: | |
| prompt: Natural language query about KEGG data (e.g., "Find human pathways related to glycolysis") | |
| endpoint: Direct KEGG API endpoint to query | |
| verbose: Whether to return detailed results | |
| Returns: | |
| Dictionary containing both the structured query and the KEGG results | |
| """ | |
| base_url = "https://rest.kegg.jp" | |
| if not prompt and not endpoint: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| if prompt: | |
| # Load schema from pickle file | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "kegg.pkl") | |
| with open(schema_path, "rb") as f: | |
| kegg_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a bioinformatics expert that helps convert natural language queries into KEGG API requests. | |
| Based on the user's natural language request, you will generate a structured query for the KEGG API. | |
| The KEGG API has the following general form: | |
| https://rest.kegg.jp/<operation>/<argument>[/<argument2>[/<argument3> ...]] | |
| Where <operation> can be one of: info, list, find, get, conv, link, ddi | |
| Here is the schema of available operations, databases, and other details: | |
| {schema} | |
| Output only a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including the base URL "https://rest.kegg.jp") | |
| 2. "description": A brief description of what the query is doing | |
| IMPORTANT: Your response must ONLY contain a JSON object with the required fields. | |
| EXAMPLES OF CORRECT OUTPUTS: | |
| - For "Find information about glycolysis pathway": {{"full_url": "https://rest.kegg.jp/info/pathway/hsa00010", "description": "Finding information about the glycolysis pathway"}} | |
| - For "Get information about the human BRCA1 gene": {{"full_url": "https://rest.kegg.jp/get/hsa:672", "description": "Retrieving information about BRCA1 gene in human"}} | |
| - For "List all human pathways": {{"full_url": "https://rest.kegg.jp/list/pathway/hsa", "description": "Listing all human-specific pathways"}} | |
| - For "Convert NCBI gene ID 672 to KEGG ID": {{"full_url": "https://rest.kegg.jp/conv/genes/ncbi-geneid:672", "description": "Converting NCBI Gene ID 672 to KEGG gene identifier"}} | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=kegg_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Extract the query info from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info["full_url"] | |
| description = query_info["description"] | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| if endpoint: | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = "Direct query to KEGG API" | |
| # Execute the KEGG API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_stringdb(prompt: str = None, endpoint: str = None, api_key: str = None, download_image: bool = False, output_dir: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the STRING protein interaction database using natural language or direct endpoint. | |
| Args: | |
| prompt: Natural language query about protein interactions | |
| endpoint: Full URL to query directly (overrides prompt) | |
| api_key: Anthropic API key for processing | |
| model: Model to use for natural language processing | |
| download_image: Whether to download image results | |
| output_dir: Directory to save downloaded files | |
| verbose: Whether to return detailed results | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_stringdb("Show protein interactions for BRCA1 and BRCA2 in humans") | |
| - Direct endpoint: query_stringdb(endpoint="https://string-db.org/api/json/network?identifiers=BRCA1,BRCA2&species=9606") | |
| """ | |
| # Base URL for STRING API | |
| base_url = "https://version-12-0.string-db.org/api" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load STRING schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "stringdb.pkl") | |
| with open(schema_path, "rb") as f: | |
| stringdb_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a protein interaction expert specialized in using the STRING database API. | |
| Based on the user's natural language request, determine the appropriate STRING API endpoint and parameters. | |
| STRING API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including all parameters) | |
| 2. "description": A brief description of what the query is doing | |
| 3. "output_format": The format of the output (json, tsv, image, svg) | |
| SPECIAL NOTES: | |
| - Common species IDs: 9606 (human), 10090 (mouse), 7227 (fruit fly), 4932 (yeast) | |
| - For protein identifiers, use either gene names (e.g., "BRCA1") or UniProt IDs (e.g., "P38398") | |
| - The "required_score" parameter accepts values from 0 to 1000 (higher means more stringent) | |
| - Add "caller_identity=bioagentos_api" as a parameter | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=stringdb_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| output_format = query_info.get("output_format", "json") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Use direct endpoint | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = "Direct query to STRING API" | |
| output_format = "json" | |
| # Try to determine output format from URL | |
| if "image" in endpoint or "svg" in endpoint: | |
| output_format = "image" | |
| # Check if we're dealing with an image request | |
| is_image = output_format in ["image", "highres_image", "svg"] | |
| if is_image: | |
| if download_image: | |
| # For images, we need to handle the download manually | |
| try: | |
| response = requests.get(endpoint, stream=True) | |
| response.raise_for_status() | |
| # Create output directory if needed | |
| if not output_dir: | |
| output_dir = "." | |
| os.makedirs(output_dir, exist_ok=True) | |
| # Generate filename based on endpoint | |
| endpoint_parts = endpoint.split("/") | |
| filename = f"string_{endpoint_parts[-2]}_{int(time.time())}.{output_format}" | |
| file_path = os.path.join(output_dir, filename) | |
| # Save the image | |
| with open(file_path, 'wb') as f: | |
| for chunk in response.iter_content(chunk_size=1024): | |
| if chunk: | |
| f.write(chunk) | |
| return { | |
| "success": True, | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "description": description, | |
| "output_format": output_format | |
| }, | |
| "result": { | |
| "image_saved": True, | |
| "file_path": file_path, | |
| "content_type": response.headers.get('Content-Type') | |
| } | |
| } | |
| except Exception as e: | |
| return { | |
| "success": False, | |
| "error": f"Error downloading image: {str(e)}", | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "description": description | |
| } | |
| } | |
| else: | |
| # Just report that an image is available but not downloaded | |
| return { | |
| "success": True, | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "description": description, | |
| "output_format": output_format | |
| }, | |
| "result": { | |
| "image_available": True, | |
| "download_url": endpoint, | |
| "note": "Set download_image=True to save the image" | |
| } | |
| } | |
| # For non-image requests, use the REST API helper | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_paleobiology(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the Paleobiology Database (PBDB) API using natural language or a direct endpoint. | |
| Args: | |
| prompt (str, required): Natural language query about fossil records | |
| endpoint (str, optional): API endpoint name or full URL | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_paleobiology("Find fossil records of Tyrannosaurus rex") | |
| - Direct endpoint: query_paleobiology(endpoint="data1.2/taxa/list.json?name=Tyrannosaurus") | |
| """ | |
| # Base URL for PBDB API | |
| base_url = "https://paleobiodb.org/data1.2" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load PBDB schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "paleobiology.pkl") | |
| with open(schema_path, "rb") as f: | |
| pbdb_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a paleobiology expert specialized in using the Paleobiology Database (PBDB) API. | |
| Based on the user's natural language request, determine the appropriate PBDB API endpoint and parameters. | |
| PBDB API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including the base URL "https://paleobiodb.org/data1.2" and format extension) | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - For taxonomic queries, be specific about taxonomic ranks and names | |
| - For geographic queries, use standard country/continent names or coordinate bounding boxes | |
| - For time interval queries, use standard geological time names (e.g., "Cretaceous", "Maastrichtian") | |
| - Use appropriate format extension (.json, .txt, .csv, .tsv) based on the query | |
| - If appropriate, use "vocab=pbdb" (default) or "vocab=com" (compact) parameter in the URL | |
| - For detailed occurrence data, include "show=paleoloc,phylo" in the parameters | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=pbdb_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| if not endpoint.startswith("http"): | |
| # Add base URL if it's just a path | |
| if not endpoint.startswith('/'): | |
| endpoint = f"{base_url}/{endpoint}" | |
| else: | |
| endpoint = f"{base_url}{endpoint}" | |
| description = "Direct query to PBDB API" | |
| # Check if we're dealing with an image request | |
| is_image = endpoint.endswith('.png') | |
| if is_image: | |
| # For image queries, we need special handling | |
| try: | |
| response = requests.get(endpoint) | |
| response.raise_for_status() | |
| # Return image metadata without the binary data | |
| return { | |
| "success": True, | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "description": description, | |
| "format": "png" | |
| }, | |
| "result": { | |
| "content_type": response.headers.get('Content-Type'), | |
| "size_bytes": len(response.content), | |
| "note": "Binary image data not included in response" | |
| } | |
| } | |
| except Exception as e: | |
| return { | |
| "success": False, | |
| "error": f"Error retrieving image: {str(e)}", | |
| "query_info": { | |
| "endpoint": endpoint, | |
| "description": description | |
| } | |
| } | |
| # For non-image requests, use the REST API helper | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_jaspar(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the JASPAR REST API using natural language or a direct endpoint. | |
| Args: | |
| prompt (str, required): Natural language query about transcription factor binding profiles | |
| endpoint (str, optional): API endpoint path (e.g., "/matrix/MA0002.2/") or full URL | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_jaspar("Find all transcription factor matrices for human") | |
| - Direct endpoint: query_jaspar(endpoint="/matrix/MA0002.2/") | |
| """ | |
| # Base URL for JASPAR API | |
| base_url = "https://jaspar.elixir.no/api/v1" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load JASPAR schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "jaspar.pkl") | |
| with open(schema_path, "rb") as f: | |
| jaspar_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a transcription factor binding site expert specialized in using the JASPAR REST API. | |
| Based on the user's natural language request, determine the appropriate JASPAR REST API endpoint and parameters. | |
| JASPAR REST API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including the base URL "https://jaspar.elixir.no/api/v1" and any parameters) | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - Common taxonomic groups include: vertebrates, plants, fungi, insects, nematodes, urochordates | |
| - Common collections include: CORE, UNVALIDATED, PENDING, etc. | |
| - Matrix IDs follow the format MA####.# (e.g., MA0002.2) | |
| - For inferring matrices from sequences, provide the protein sequence directly in the path | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=jaspar_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| if not endpoint.startswith("http"): | |
| # Clean up endpoint format | |
| if not endpoint.startswith("/"): | |
| endpoint = "/" + endpoint | |
| # Ensure endpoint ends with / | |
| if not endpoint.endswith("/"): | |
| endpoint = endpoint + "/" | |
| # Add base URL | |
| endpoint = f"{base_url}{endpoint}" | |
| description = "Direct query to JASPAR API" | |
| # Execute the JASPAR API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_worms(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the World Register of Marine Species (WoRMS) REST API using natural language or a direct endpoint. | |
| Args: | |
| prompt (str, required): Natural language query about marine species | |
| endpoint (str, optional): Full URL or endpoint specification | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_worms("Find information about the blue whale") | |
| - Direct endpoint: query_worms(endpoint="https://www.marinespecies.org/rest/AphiaRecordByName/Balaenoptera%20musculus") | |
| """ | |
| # Base URL for WoRMS API | |
| base_url = "https://www.marinespecies.org/rest" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load WoRMS schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "worms.pkl") | |
| with open(schema_path, "rb") as f: | |
| worms_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a marine biology expert specialized in using the World Register of Marine Species (WoRMS) API. | |
| Based on the user's natural language request, determine the appropriate WoRMS API endpoint and parameters. | |
| WORMS API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including the base URL "https://www.marinespecies.org/rest" and any path/query parameters) | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - For taxonomic searches, be precise with scientific names and use proper capitalization | |
| - For fuzzy matching, include "fuzzy=true" in the URL query parameters | |
| - When searching by name, prefer "AphiaRecordByName" for exact matches and "AphiaRecordsByName" for broader results | |
| - AphiaID is the main identifier in WoRMS (e.g., Blue Whale is 137087) | |
| - For multiple IDs or names, use the appropriate POST endpoint | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=worms_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL and details from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| if not endpoint.startswith("http"): | |
| # Add base URL if it's just a path | |
| if not endpoint.startswith('/'): | |
| endpoint = f"{base_url}/{endpoint}" | |
| else: | |
| endpoint = f"{base_url}{endpoint}" | |
| description = "Direct query to WoRMS API" | |
| # Execute the WoRMS API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method='GET', | |
| description=description | |
| ) | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_cbioportal(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the cBioPortal REST API using natural language or a direct endpoint. | |
| Args: | |
| prompt (str, required): Natural language query about cancer genomics data | |
| endpoint (str, optional): API endpoint path (e.g., "/studies/brca_tcga/patients") or full URL | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_cbioportal("Find mutations in BRCA1 for breast cancer") | |
| - Direct endpoint: query_cbioportal(endpoint="/studies/brca_tcga/molecular-profiles") | |
| """ | |
| # Base URL for cBioPortal API | |
| base_url = "https://www.cbioportal.org/api" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load cBioPortal schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "cbioportal.pkl") | |
| with open(schema_path, "rb") as f: | |
| cbioportal_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a cancer genomics expert specialized in using the cBioPortal REST API. | |
| Based on the user's natural language request, determine the appropriate cBioPortal REST API endpoint and parameters. | |
| CBIOPORTAL REST API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including the base URL "https://www.cbioportal.org/api" and any parameters) | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - For gene queries, use either Hugo symbol (e.g., "BRCA1") or Entrez ID (e.g., 672) | |
| - For pagination, include parameters "pageNumber" and "pageSize" if needed | |
| - For mutation data queries, always include appropriate sample identifiers | |
| - Common studies include: "brca_tcga" (breast cancer), "gbm_tcga" (glioblastoma), "luad_tcga" (lung adenocarcinoma) | |
| - For molecular profiles, common IDs follow pattern: "[study]_[data_type]" (e.g., "brca_tcga_mutations") | |
| - Consider including "projection=DETAILED" for more comprehensive results when appropriate | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=cbioportal_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| if not endpoint.startswith("http"): | |
| # Clean up endpoint format | |
| if not endpoint.startswith("/"): | |
| endpoint = "/" + endpoint | |
| # Add base URL | |
| endpoint = f"{base_url}{endpoint}" | |
| description = "Direct query to cBioPortal API" | |
| # Execute the cBioPortal API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_clinvar(prompt: str = None, search_term: str = None, max_results: int = 3) -> dict: | |
| """ | |
| Take a natural language prompt and convert it to a structured ClinVar query. | |
| Args: | |
| prompt: Natural language query about genetic variants (e.g., "Find pathogenic BRCA1 variants") | |
| search_term: Direct search term for ClinVar | |
| max_results: Maximum number of results to return | |
| Returns: | |
| Dictionary containing both the structured query and the ClinVar results | |
| """ | |
| if not prompt and not search_term: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| if prompt: | |
| # Load ClinVar schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "clinvar.pkl") | |
| with open(schema_path, "rb") as f: | |
| clinvar_schema = pickle.load(f) | |
| # ClinVar system prompt template | |
| system_prompt_template = """ | |
| You are a genetics research assistant that helps convert natural language queries into structured ClinVar search queries. | |
| Based on the user's natural language request, you will generate a structured search for the ClinVar database. | |
| Output only a JSON object with the following fields: | |
| 1. "search_term": The exact search query to use with the ClinVar API | |
| IMPORTANT: Your response must ONLY contain a JSON object with the search term field. | |
| Your "search_term" MUST strictly follow these ClinVar search syntax rules/tags: | |
| {schema} | |
| For combining terms: Use AND, OR, NOT (must be capitalized) | |
| For complex logic: Use parentheses | |
| For terms with multiple words: use double quotes escaped with a backslash or underscore (e.g. breast_cancer[dis] or \"breast cancer\"[dis]) | |
| Example: "BRCA1[gene] AND (pathogenic[clinsig] OR likely_pathogenic[clinsig])" | |
| EXAMPLES OF CORRECT QUERIES: | |
| - For "pathogenic BRCA1 variants": "BRCA1[gene] AND clinsig_pathogenic[prop]" | |
| - For "Specific RS": "rs6025[rsid]" | |
| - For "Combined search with multiple criteria": "BRCA1[gene] AND origin_germline[prop]" | |
| - For "Find variants in a specific genomic region": "17[chr] AND 43000000:44000000[chrpos37]" | |
| - If query asks for pathogenicity of a variant, it's asking for all possible germline classifications of the variant, so just [gene] AND [variant] is needed | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=clinvar_schema, | |
| system_template=system_prompt_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| search_term = query_info.get("search_term", "") | |
| if not search_term: | |
| return { | |
| "error": "Failed to generate a valid search term from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| return _query_ncbi_database( | |
| database="clinvar", | |
| search_term=search_term, | |
| max_results=max_results, | |
| ) | |
| def query_geo(prompt: str = None, search_term: str = None, max_results: int = 3) -> dict: | |
| """ | |
| Query the NCBI Gene Expression Omnibus (GEO) using natural language or a direct search term. | |
| Args: | |
| prompt: Natural language query about RNA-seq, microarray, or other expression data | |
| search_term: Direct search term in GEO syntax | |
| max_results: Maximum number of results to return | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_geo("Find RNA-seq datasets for breast cancer") | |
| - Direct search: query_geo(search_term="RNA-seq AND breast cancer AND gse[ETYP]") | |
| """ | |
| if not prompt and not search_term: | |
| return {"error": "Either a prompt or a search term must be provided"} | |
| database = "gds" # Default database | |
| if prompt: | |
| # Load GEO schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "geo.pkl") | |
| with open(schema_path, "rb") as f: | |
| geo_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a bioinformatics research assistant that helps convert natural language queries into structured GEO (Gene Expression Omnibus) search queries. | |
| Based on the user's natural language request, you will generate a structured search for the GEO database. | |
| Output only a JSON object with the following fields: | |
| 1. "search_term": The exact search query to use with the GEO API | |
| 2. "database": The specific GEO database to search (either "gds" for GEO DataSets or "geoprofiles" for GEO Profiles) | |
| IMPORTANT: Your response must ONLY contain a JSON object with the required fields. | |
| Your "search_term" MUST strictly follow these GEO search syntax rules/tags: | |
| {schema} | |
| For combining terms: Use AND, OR, NOT (must be capitalized) | |
| For complex logic: Use parentheses | |
| For terms with multiple words: use double quotes or underscore (e.g. "breast cancer"[Title]) | |
| Date ranges use colon format: 2015/01:2020/12[PDAT] | |
| Choose the appropriate database based on the user's query: | |
| - gds: GEO DataSets (contains Series, Datasets, Platforms, Samples metadata) | |
| - geoprofiles: GEO Profiles (contains gene expression data) | |
| If database isn't clearly specified, default to "gds" as it contains most common experiment metadata. | |
| EXAMPLES OF CORRECT OUTPUTS: | |
| - For "RNA-seq data in breast cancer": {"search_term": "RNA-seq AND breast cancer AND gse[ETYP]", "database": "gds"} | |
| - For "Mouse microarray data from 2020": {"search_term": "Mus musculus[ORGN] AND 2020[PDAT] AND microarray AND gse[ETYP]", "database": "gds"} | |
| - For "Expression profiles of TP53 in lung cancer": {"search_term": "TP53[Gene Symbol] AND lung cancer", "database": "geoprofiles"} | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=geo_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the search term and database from Gemini's response | |
| query_info = gemini_result["data"] | |
| search_term = query_info.get("search_term", "") | |
| database = query_info.get("database", "gds") | |
| if not search_term: | |
| return { | |
| "error": "Failed to generate a valid search term from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| # Execute the GEO query using the helper function | |
| result = _query_ncbi_database( | |
| database=database, | |
| search_term=search_term, | |
| max_results=max_results, | |
| ) | |
| return result | |
| def query_dbsnp(prompt: str = None, search_term: str = None, max_results: int = 3) -> dict: | |
| """ | |
| Query the NCBI dbSNP database using natural language or a direct search term. | |
| Args: | |
| prompt: Natural language query about genetic variants/SNPs | |
| search_term: Direct search term in dbSNP syntax | |
| max_results: Maximum number of results to return | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_dbsnp("Find pathogenic variants in BRCA1") | |
| - Direct search: query_dbsnp(search_term="BRCA1[Gene Name] AND pathogenic[Clinical Significance]") | |
| """ | |
| if not prompt and not search_term: | |
| return {"error": "Either a prompt or a search term must be provided"} | |
| if prompt: | |
| # Load dbSNP schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "dbsnp.pkl") | |
| with open(schema_path, "rb") as f: | |
| dbsnp_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a genetics research assistant that helps convert natural language queries into structured dbSNP search queries. | |
| Based on the user's natural language request, you will generate a structured search for the dbSNP database. | |
| Output only a JSON object with the following fields: | |
| 1. "search_term": The exact search query to use with the dbSNP API | |
| IMPORTANT: Your response must ONLY contain a JSON object with the search term field. | |
| Your "search_term" MUST strictly follow these dbSNP search syntax rules/tags: | |
| {schema} | |
| For combining terms: Use AND, OR, NOT (must be capitalized) | |
| For complex logic: Use parentheses | |
| For terms with multiple words: use double quotes (e.g. "breast cancer"[Disease Name]) | |
| EXAMPLES OF CORRECT QUERIES: | |
| - For "pathogenic variants in BRCA1": "BRCA1[Gene Name] AND pathogenic[Clinical Significance]" | |
| - For "specific SNP rs6025": "rs6025[rs]" | |
| - For "SNPs in a genomic region": "17[Chromosome] AND 41196312:41277500[Base Position]" | |
| - For "common SNPs in EGFR": "EGFR[Gene Name] AND common[COMMON]" | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=dbsnp_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the search term from Gemini's response | |
| query_info = gemini_result["data"] | |
| search_term = query_info.get("search_term", "") | |
| if not search_term: | |
| return { | |
| "error": "Failed to generate a valid search term from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| # Execute the dbSNP query using the helper function | |
| result = _query_ncbi_database( | |
| database="snp", | |
| search_term=search_term, | |
| max_results=max_results, | |
| ) | |
| return result | |
| def query_ucsc(prompt: str = None, endpoint: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the UCSC Genome Browser API using natural language or a direct endpoint. | |
| Args: | |
| prompt: Natural language query about genomic data | |
| endpoint: Full URL or endpoint specification with parameters | |
| verbose: Whether to return detailed results | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_ucsc("Get DNA sequence of chromosome M positions 1-100 in human genome") | |
| - Direct endpoint: query_ucsc(endpoint="https://api.genome.ucsc.edu/getData/sequence?genome=hg38&chrom=chrM&start=1&end=100") | |
| """ | |
| # Base URL for UCSC API | |
| base_url = "https://api.genome.ucsc.edu" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load UCSC schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "ucsc.pkl") | |
| with open(schema_path, "rb") as f: | |
| ucsc_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a genomics expert specialized in using the UCSC Genome Browser API. | |
| Based on the user's natural language request, determine the appropriate UCSC Genome Browser API endpoint and parameters. | |
| UCSC GENOME BROWSER API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "full_url": The complete URL to query (including the base URL "https://api.genome.ucsc.edu" and all parameters) | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - For chromosome names, always include the "chr" prefix (e.g., "chr1", "chrX", "chrM") | |
| - Genomic positions are 0-based (first base is position 0) | |
| - For "start" and "end" parameters, both must be provided together | |
| - The "maxItemsOutput" parameter can be used to limit the amount of data returned | |
| - Common genomes include: "hg38" (human), "mm39" (mouse), "danRer11" (zebrafish) | |
| - For sequence data, use "getData/sequence" endpoint | |
| - For chromosome listings, use "list/chromosomes" endpoint | |
| - For available genomes, use "list/ucscGenomes" endpoint | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=ucsc_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the full URL from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("full_url", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| if not endpoint.startswith("http"): | |
| # Add base URL if it's just a path | |
| endpoint = f"{base_url}/{endpoint}" | |
| description = "Direct query to UCSC Genome Browser API" | |
| # Execute the UCSC API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| # Format the results if successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_ensembl(prompt: str = None, endpoint: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the Ensembl REST API using natural language or a direct endpoint. | |
| Args: | |
| prompt: Natural language query about genomic data | |
| endpoint: Direct API endpoint to query (e.g., "lookup/symbol/human/BRCA2") or full URL | |
| verbose: Whether to return detailed results | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_ensembl("Get information about the human BRCA2 gene") | |
| - Direct endpoint: query_ensembl(endpoint="lookup/symbol/homo_sapiens/BRCA2") | |
| """ | |
| print("IN QUERY ENSEMBL") | |
| print("PROMPT: ", prompt) | |
| print("ENDPOINT: ", endpoint) | |
| # Base URL for Ensembl API | |
| base_url = "https://rest.ensembl.org" | |
| # Ensure we have either a prompt or an endpoint | |
| if not prompt and not endpoint: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load Ensembl schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "ensembl.pkl") | |
| with open(schema_path, "rb") as f: | |
| ensembl_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a genomics and bioinformatics expert specialized in using the Ensembl REST API. | |
| Based on the user's natural language request, determine the appropriate Ensembl REST API endpoint and parameters. | |
| ENSEMBL REST API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The API endpoint to query (e.g., "lookup/symbol/homo_sapiens/BRCA2") | |
| 2. "params": An object containing query parameters specific to the endpoint | |
| 3. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - Chromosome region queries have a maximum length of 4900000 bp inclusive, so bp of start and end should be 4900000 bp apart. If the user's query exceeds this limit, Ensembl will return an error. | |
| - For symbol lookups, the format is "lookup/symbol/[species]/[symbol]" | |
| - To find the coordinates of a band on a chromosome, use /info/assembly/homo_sapiens/[chromosome] with parameters "band":1 | |
| - To find the overlapping genes of a genomic region, use /overlap/region/homo_sapiens/[chromosome]:[start]-[end] | |
| - For sequence queries, specify the sequence type in parameters (genomic, cdna, cds, protein) | |
| - For converting rsID to hg38 genomic coordinates, use the "GET id/variation/[species]/[rsid]" endpoint | |
| - Many endpoints support "content-type" parameter for format specification (application/json, text/xml) | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=ensembl_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| params = query_info.get("params", {}) | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| if endpoint.startswith("http"): | |
| # If a full URL is provided, extract the endpoint part | |
| if endpoint.startswith(base_url): | |
| endpoint = endpoint[len(base_url):].lstrip('/') | |
| params = {} | |
| description = "Direct query to Ensembl API" | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = endpoint[1:] | |
| # Prepare headers for JSON response | |
| headers = { | |
| "Content-Type": "application/json", | |
| "Accept": "application/json" | |
| } | |
| # Construct the URL | |
| url = f"{base_url}/{endpoint}" | |
| # Execute the Ensembl API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=url, | |
| method="GET", | |
| params=params, | |
| headers=headers, | |
| description=description | |
| ) | |
| # Format the results if successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_opentarget_genetics(prompt: str = None, query: str = None, variables: dict = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the OpenTargets Genetics API using natural language or a direct GraphQL query. | |
| Args: | |
| prompt (str, required): Natural language query about genetic targets and variants | |
| query (str, optional): Direct GraphQL query string | |
| variables (dict, optional): Variables for the GraphQL query | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_opentarget("Get information about variant 1_154453788_C_T") | |
| - Direct query: query_opentarget(query="query variantInfo($variantId: String!) {...}", | |
| variables={"variantId": "1_154453788_C_T"}) | |
| """ | |
| # Constants and initialization | |
| OPENTARGETS_URL = "https://api.genetics.opentargets.org/graphql" | |
| # Ensure we have either a prompt or a query | |
| if prompt is None and query is None: | |
| return {"error": "Either a prompt or a GraphQL query must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load OpenTargets schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "opentarget_genetics.pkl") | |
| with open(schema_path, "rb") as f: | |
| opentarget_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are an expert in translating natural language requests into GraphQL queries for the OpenTargets Genetics API. | |
| Here is a schema of the main types and queries available in the OpenTargets Genetics API: | |
| {schema} | |
| Translate the user's natural language request into a valid GraphQL query for this API. | |
| Return only a JSON object with two fields: | |
| 1. "query": The complete GraphQL query string | |
| 2. "variables": A JSON object containing the variables needed for the query | |
| SPECIAL NOTES: | |
| - Variant IDs are typically in the format 'chromosome_position_ref_alt' (e.g., '1_154453788_C_T') | |
| - For L2G (locus-to-gene) queries, you need both a variant ID and a study ID | |
| - The API can provide variant information, QTLs, PheWAS results, pathogenicity scores, etc. | |
| - For mutations by gene, use the approved gene symbol (e.g., "BRCA1") | |
| - Always escape special characters, including quotes, in the query string (eg. \" instead of ") | |
| Return ONLY the JSON object with no additional text or explanations. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=opentarget_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the query and variables from Gemini's response | |
| query_info = gemini_result["data"] | |
| query = query_info.get("query", "") | |
| if variables is None: # Only use Claude's variables if none provided | |
| variables = query_info.get("variables", {}) | |
| if not query: | |
| return { | |
| "error": "Failed to generate a valid GraphQL query from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| # Execute the GraphQL query | |
| api_result = _query_rest_api( | |
| endpoint=OPENTARGETS_URL, | |
| method="POST", | |
| json_data={"query": query, "variables": variables or {}}, | |
| headers={"Content-Type": "application/json"} | |
| ) | |
| if not api_result["success"]: | |
| return api_result | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_opentarget(prompt: str = None, query: str = None, variables: dict = None, verbose: bool = False) -> dict: | |
| """ | |
| Query the OpenTargets Platform API using natural language or a direct GraphQL query. | |
| Args: | |
| prompt: Natural language query about drug targets, diseases, and mechanisms | |
| query: Direct GraphQL query string | |
| variables: Variables for the GraphQL query | |
| verbose: Whether to return detailed results | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_opentarget("Find drug targets for Alzheimer's disease") | |
| - Direct query: query_opentarget(query="query diseaseAssociations($diseaseId: String!) {...}", | |
| variables={"diseaseId": "EFO_0000249"}) | |
| """ | |
| # Constants and initialization | |
| OPENTARGETS_URL = "https://api.platform.opentargets.org/api/v4/graphql" | |
| # Ensure we have either a prompt or a query | |
| if prompt is None and query is None: | |
| return {"error": "Either a prompt or a GraphQL query must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load OpenTargets schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "opentarget.pkl") | |
| with open(schema_path, "rb") as f: | |
| opentarget_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are an expert in translating natural language requests into GraphQL queries for the OpenTargets Platform API. | |
| Here is a schema of the main types and queries available in the OpenTargets Platform API: | |
| {schema} | |
| Translate the user's natural language request into a valid GraphQL query for this API. | |
| Return only a JSON object with two fields: | |
| 1. "query": The complete GraphQL query string | |
| 2. "variables": A JSON object containing the variables needed for the query | |
| SPECIAL NOTES: | |
| - Disease IDs typically use EFO ontology (e.g., "EFO_0000249" for Alzheimer's disease) | |
| - Target IDs typically use Ensembl IDs (e.g., "ENSG00000197386" for ENSG00000197386) | |
| - The API can provide information about drug-target associations, disease-target associations, etc. | |
| - Always limit results to a reasonable number using "first" parameter (e.g., first: 10) | |
| - Always escape special characters, including quotes, in the query string (eg. \\" instead of ") | |
| Return ONLY the JSON object with no additional text or explanations. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=opentarget_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the query and variables from Gemini's response | |
| query_info = gemini_result["data"] | |
| query = query_info.get("query", "") | |
| if variables is None: # Only use Claude's variables if none provided | |
| variables = query_info.get("variables", {}) | |
| if not query: | |
| return { | |
| "error": "Failed to generate a valid GraphQL query from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| # Execute the GraphQL query | |
| api_result = _query_rest_api( | |
| endpoint=OPENTARGETS_URL, | |
| method="POST", | |
| json_data={"query": query, "variables": variables or {}}, | |
| headers={"Content-Type": "application/json"}, | |
| description="OpenTargets Platform GraphQL query" | |
| ) | |
| # Format the results if not verbose and successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| api_result["result"] = _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_gwas_catalog(prompt: str = None, endpoint: str = None, max_results: int = 3) -> dict: | |
| """ | |
| Query the GWAS Catalog API using natural language or a direct endpoint. | |
| Args: | |
| prompt: Natural language query about GWAS data | |
| endpoint: Full API endpoint to query (e.g., "https://www.ebi.ac.uk/gwas/rest/api/studies?diseaseTraitId=EFO_0001360") | |
| max_results: Maximum number of results to return | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_gwas_catalog("Find GWAS studies related to Type 2 diabetes") | |
| - Direct endpoint: query_gwas_catalog(endpoint="studies", params={"diseaseTraitId": "EFO_0001360"}) | |
| """ | |
| # Base URL for GWAS Catalog API | |
| base_url = "https://www.ebi.ac.uk/gwas/rest/api" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load GWAS Catalog schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "gwas_catalog.pkl") | |
| with open(schema_path, "rb") as f: | |
| gwas_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a genomics expert specialized in using the GWAS Catalog API. | |
| Based on the user's natural language request, determine the appropriate GWAS Catalog API endpoint and parameters. | |
| GWAS CATALOG API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The API endpoint to query (e.g., "studies", "associations") | |
| 2. "params": An object containing query parameters specific to the endpoint | |
| 3. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - For disease/trait searches, consider using the "EFO" identifiers when possible | |
| - Common endpoints include: "studies", "associations", "singleNucleotidePolymorphisms", "efoTraits" | |
| - For pagination, use "size" and "page" parameters | |
| - For filtering by p-value, use "pvalueMax" parameter | |
| - GWAS Catalog uses a HAL-based REST API | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=gwas_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| params = query_info.get("params", {}) | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| if endpoint is None: | |
| endpoint = "" # Use root endpoint | |
| params = {"size": max_results} | |
| description = f"Direct query to {endpoint}" | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = endpoint[1:] | |
| # Construct the URL | |
| url = f"{base_url}/{endpoint}" | |
| # Execute the GWAS Catalog API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=url, | |
| method="GET", | |
| params=params, | |
| description=description | |
| ) | |
| return api_result | |
| def query_gnomad(prompt: str = None, gene_symbol: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query gnomAD for variants in a gene using natural language or direct gene symbol. | |
| Args: | |
| prompt: Natural language query about genetic variants | |
| gene_symbol: Gene symbol (e.g., "BRCA1") | |
| verbose: Whether to return detailed results | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Direct gene: query_gnomad(gene_symbol="BRCA1") | |
| - Natural language: query_gnomad(prompt="Find variants in the TP53 gene") | |
| """ | |
| # Base URL for gnomAD API | |
| base_url = "https://gnomad.broadinstitute.org/api" | |
| # Ensure we have either a prompt or a gene_symbol | |
| if prompt is None and gene_symbol is None: | |
| return {"error": "Either a prompt or a gene_symbol must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt and not gene_symbol: | |
| # Load gnomAD schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "gnomad.pkl") | |
| with open(schema_path, "rb") as f: | |
| gnomad_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a genomics expert specialized in using the gnomAD GraphQL API. | |
| Based on the user's natural language request, extract the gene symbol and relevant parameters and create the gnomAD GraphQL query. | |
| GnomAD GraphQL API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "query": The complete GraphQL query string | |
| SPECIAL NOTES: | |
| - The gene_symbol should be the official gene symbol (e.g., "BRCA1" not "breast cancer gene 1") | |
| - If no reference genome is specified, default to GRCh38 | |
| - If no dataset is specified, default to gnomad_r4 | |
| - Return only a single gene symbol, even if multiple are mentioned | |
| - Always escape special characters, including quotes, in the query string (eg. \" instead of ") | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=gnomad_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the gene symbol from Gemini's response | |
| query_info = gemini_result["data"] | |
| query_str = query_info.get("query", "") | |
| if not query_str: | |
| return { | |
| "error": "Failed to extract a valid query from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Load gnomAD schema for gene_symbol substitution | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "gnomad.pkl") | |
| with open(schema_path, "rb") as f: | |
| gnomad_schema = pickle.load(f) | |
| description = f"Query gnomAD for variants in {gene_symbol}" | |
| # replace BRCA1 with gene_symbol | |
| query_str = gnomad_schema.replace("BRCA1", gene_symbol) | |
| api_result = _query_rest_api( | |
| endpoint=base_url, | |
| method="POST", | |
| json_data={"query": query_str}, | |
| headers={"Content-Type": "application/json"}, | |
| description=description | |
| ) | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def blast_sequence(sequence: str, database: str, program: str) -> Union[Dict[str, Union[str, float]], str]: | |
| """ | |
| Identifies a DNA sequence using NCBI BLAST with improved error handling, timeout management, and debugging | |
| Args: | |
| sequence: The sequence to identify. If DNA, use database: core_nt, program: blastn; | |
| if protein, use database: nr, program: blastp | |
| database: The BLAST database to search against | |
| program: The BLAST program to use | |
| Returns: | |
| A dictionary containing the title, e-value, identity percentage, and coverage percentage of the best alignment | |
| """ | |
| max_attempts = 1 # One initial attempt plus one retry | |
| attempts = 0 | |
| max_runtime = 600 # 10 minutes in seconds | |
| while attempts < max_attempts: | |
| try: | |
| attempts += 1 | |
| query_sequence = Seq(sequence) | |
| # Start timer | |
| start_time = time.time() | |
| # Submit BLAST job | |
| print(f"Submitting BLAST job (attempt {attempts}/{max_attempts})...") | |
| result_handle = NCBIWWW.qblast(program, database, query_sequence, expect=100, word_size=7, megablast=True) | |
| # Parse results with timeout check | |
| blast_records = NCBIXML.parse(result_handle) | |
| blast_record = None | |
| # Try to get the first record with timeout check | |
| while time.time() - start_time < max_runtime: | |
| try: | |
| # Set a short timeout for next operation | |
| blast_record = next(blast_records) # Get first record | |
| break # Successfully got the record | |
| except StopIteration: | |
| # No more records | |
| return "No BLAST results found" | |
| except Exception as e: | |
| # Check if we've exceeded the time limit | |
| if time.time() - start_time >= max_runtime: | |
| if attempts < max_attempts: | |
| print("BLAST job timeout exceeded. Resubmitting...") | |
| break # Break to retry | |
| else: | |
| return "BLAST search failed after maximum attempts due to timeout" | |
| # Brief pause before trying again | |
| time.sleep(1) | |
| # Check if we timed out during record retrieval | |
| if blast_record is None: | |
| if attempts < max_attempts: | |
| continue # Retry | |
| else: | |
| return "BLAST search failed after maximum attempts due to timeout" | |
| # Debug information | |
| print(f"Number of alignments found: {len(blast_record.alignments)}") | |
| if blast_record.alignments: | |
| for alignment in blast_record.alignments: | |
| print("\nAlignment:") | |
| print(f"hit_id: {alignment.hit_id}") | |
| print(f"hit_def: {alignment.hit_def}") | |
| print(f"accession: {alignment.accession}") | |
| for hsp in alignment.hsps: | |
| print(f"E-value: {hsp.expect}") | |
| print(f"Score: {hsp.score}") | |
| print(f"Identities: {hsp.identities}/{hsp.align_length}") | |
| return { | |
| 'hit_id': alignment.hit_id, | |
| 'hit_def': alignment.hit_def, | |
| 'accession': alignment.accession, | |
| 'e_value': hsp.expect, | |
| 'identity': (hsp.identities / float(hsp.align_length)) * 100, | |
| 'coverage': len(hsp.query) / len(sequence) * 100 | |
| } | |
| else: | |
| return "No alignments found - sequence might be too short or low complexity" | |
| except Exception as e: | |
| if attempts < max_attempts: | |
| print(f"Error during BLAST search: {str(e)}. Retrying...") | |
| time.sleep(2) # Wait briefly before retrying | |
| else: | |
| return f"Error during BLAST search after maximum attempts: {str(e)}" | |
| return "BLAST search failed after maximum attempts" | |
| def query_reactome(prompt: str = None, endpoint: str = None, download: bool = False, output_dir: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the Reactome database using natural language or a direct endpoint. | |
| Args: | |
| prompt: Natural language query about biological pathways | |
| endpoint: Direct API endpoint or full URL | |
| download: Whether to download pathway diagrams | |
| output_dir: Directory to save downloaded files | |
| verbose: Whether to return detailed results | |
| Returns: | |
| Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_reactome("Find pathways related to DNA repair") | |
| - Direct endpoint: query_reactome(endpoint="data/pathways/R-HSA-73894") | |
| """ | |
| # Base URLs for Reactome APIs | |
| content_base_url = "https://reactome.org/ContentService" | |
| analysis_base_url = "https://reactome.org/AnalysisService" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # Create output directory if downloading and directory doesn't exist | |
| if download and output_dir: | |
| os.makedirs(output_dir, exist_ok=True) | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load Reactome schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "reactome.pkl") | |
| with open(schema_path, "rb") as f: | |
| reactome_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a bioinformatics expert specialized in using the Reactome API. | |
| Based on the user's natural language request, determine the appropriate Reactome API endpoint and parameters. | |
| REACTOME API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The API endpoint to query (e.g., "data/pathways/PATHWAY_ID", "data/query/GENE_SYMBOL") | |
| 2. "base": Which base URL to use ("content" for ContentService or "analysis" for AnalysisService) | |
| 3. "params": An object containing query parameters specific to the endpoint | |
| 4. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - Reactome has two primary APIs: ContentService (for retrieving specific pathway data) and AnalysisService (for analyzing gene lists) | |
| - For pathway queries, use "data/pathways/PATHWAY_ID" with the pathway stable identifier (e.g., R-HSA-73894) | |
| - For gene queries, use "data/query/GENE" with official gene symbol (e.g., "BRCA1") | |
| - For pathway diagrams, include "download: true" in your response if the query is for pathway visualization | |
| - Common human pathway IDs start with "R-HSA-" | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=reactome_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| base = query_info.get("base", "content") # Default to ContentService | |
| params = query_info.get("params", {}) | |
| description = query_info.get("description", "") | |
| should_download = query_info.get("download", download) # Override download if specified | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| if endpoint.startswith("http"): | |
| # Full URL already provided | |
| if "ContentService" in endpoint: | |
| base = "content" | |
| elif "AnalysisService" in endpoint: | |
| base = "analysis" | |
| else: | |
| base = "content" # Default | |
| else: | |
| # Just endpoint provided, assume ContentService by default | |
| base = "content" | |
| params = {} | |
| description = f"Direct query to Reactome {base} API: {endpoint}" | |
| should_download = download | |
| # Select base URL based on API type | |
| base_url = content_base_url if base == "content" else analysis_base_url | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = endpoint[1:] | |
| # Construct the URL | |
| if endpoint.startswith("http"): | |
| url = endpoint # Full URL already provided | |
| else: | |
| url = f"{base_url}/{endpoint}" | |
| # Execute the Reactome API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=url, | |
| method="GET", | |
| params=params, | |
| description=description | |
| ) | |
| # Handle downloading pathway diagrams if requested | |
| if should_download and api_result.get("success") and "result" in api_result: | |
| result = api_result["result"] | |
| pathway_id = None | |
| # Try to extract pathway ID from result | |
| if isinstance(result, dict): | |
| pathway_id = result.get("stId") or result.get("dbId") | |
| # If we have a pathway ID and output directory, download diagram | |
| if pathway_id and output_dir: | |
| diagram_url = f"{content_base_url}/data/pathway/{pathway_id}/diagram" | |
| try: | |
| diagram_response = requests.get(diagram_url) | |
| diagram_response.raise_for_status() | |
| # Save diagram file | |
| diagram_path = os.path.join(output_dir, f"{pathway_id}_diagram.png") | |
| with open(diagram_path, "wb") as f: | |
| f.write(diagram_response.content) | |
| api_result["diagram_path"] = diagram_path | |
| except Exception as e: | |
| api_result["diagram_error"] = f"Failed to download diagram: {str(e)}" | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| return _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_regulomedb(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = False) -> dict: | |
| """ | |
| Query the RegulomeDB database using natural language or direct variant/coordinate specification. | |
| Args: | |
| prompt (str, required): Natural language query about regulatory elements | |
| endpoint (str, optional): Direct endpoint URL or variant/coordinate specification | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_regulomedb("Find regulatory elements for rs35675666") | |
| - Direct variant: query_regulomedb(variant="rs35675666") | |
| - Coordinates: query_regulomedb(coordinates="chr11:5246919-5246919") | |
| """ | |
| # Base URL for RegulomeDB API | |
| base_url = "https://regulomedb.org/regulome-search/" | |
| # Ensure we have either a prompt, variant, or coordinates | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt, variant ID, or genomic coordinates must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt and not endpoint: | |
| # Create system prompt template | |
| system_template = """ | |
| You are a genomics expert specialized in using the RegulomeDB API. | |
| Based on the user's natural language request, extract the variant ID or genomic coordinates they want to query. | |
| Your response should be a JSON object with ONLY ONE of the following fields: | |
| 1. "endpoint": The API endpoint to query (e.g., "https://regulomedb.org/regulome-search/?regions=chr11:5246919-5246919&genome=GRCh38") | |
| SPECIAL NOTES: | |
| - RegulomeDB only works with human genome data | |
| - Variant IDs should be rsIDs from dbSNP when possible. The endpoint should be in the format https://regulomedb.org/regulome-search/?regions=rsID&genome=GRCh38 | |
| - Thumbnails for chip and chromatin should be in the format https://regulomedb.org/regulome-search?regions=chr11:5246919-5246919&genome=GRCh38/thumbnail=chip | |
| - Coordinates should be in GRCh37/hg19 format | |
| - For single base queries, use the same position for start and end (e.g., "chr11:5246919-5246919") | |
| - Chromosome should be specified with "chr" prefix (e.g., "chr11" not just "11") | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=None, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the variant or coordinates from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to extract a valid variant ID or coordinates from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| description = f"Query RegulomeDB for {endpoint}" | |
| # Construct the request URL | |
| endpoint = endpoint | |
| # Execute the RegulomeDB API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| headers = {'Accept': 'application/json'} | |
| ) | |
| # Format the results if not verbose and successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| api_result["result"] = _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_pride(prompt: str = None, endpoint: str = None, api_key: str = None, max_results: int = 3) -> dict: | |
| """ | |
| Query the PRIDE (PRoteomics IDEntifications) database using natural language or a direct endpoint. | |
| Args: | |
| prompt (str, required): Natural language query about proteomics data | |
| endpoint (str, optional): The full endpoint to query (e.g., "https://www.ebi.ac.uk/pride/ws/archive/v2/projects?keyword=breast%20cancer") | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| max_results (int): Maximum number of results to return | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_pride("Find proteomics data related to breast cancer") | |
| - Direct endpoint: query_pride(endpoint="projects", params={"keyword": "breast cancer"}) | |
| """ | |
| # Base URL for PRIDE API | |
| base_url = "https://www.ebi.ac.uk/pride/ws/archive/v2" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load PRIDE schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "pride.pkl") | |
| with open(schema_path, "rb") as f: | |
| pride_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a proteomics expert specialized in using the PRIDE API. | |
| Based on the user's natural language request, determine the appropriate PRIDE API endpoint and parameters. | |
| PRIDE API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The full url endpoint to query | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - PRIDE is a repository for proteomics data stored at EBI | |
| - Common endpoints include: "projects", "assays", "files", "proteins", "peptideevidences" | |
| - For searching projects, you can use parameters like "keyword", "species", "tissue", "disease" | |
| - For pagination, use "page" and "pageSize" parameters | |
| - Most results include PagingObject and FieldsObject structures | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=pride_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| params = query_info.get("params", {}) | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| params = {"pageSize": max_results, "page": 0} | |
| description = f"Direct query to PRIDE {endpoint}" | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = f"Direct query to provided endpoint" | |
| # Execute the PRIDE API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| params=params, | |
| description=description | |
| ) | |
| return api_result | |
| def query_gtopdb(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the Guide to PHARMACOLOGY database (GtoPdb) using natural language or a direct endpoint. | |
| Args: | |
| prompt (str, required): Natural language query about drug targets, ligands, and interactions | |
| endpoint (str, optional): Full API endpoint to query (e.g., "https://www.guidetopharmacology.org/services/targets?type=GPCR&name=beta-2") | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_gtopdb("Find ligands that target the beta-2 adrenergic receptor") | |
| - Direct endpoint: query_gtopdb(endpoint="targets", params={"type": "GPCR", "name": "beta-2"}) | |
| """ | |
| # Base URL for GtoPdb API | |
| base_url = "https://www.guidetopharmacology.org/services" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load GtoPdb schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "gtopdb.pkl") | |
| with open(schema_path, "rb") as f: | |
| gtopdb_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a pharmacology expert specialized in using the Guide to PHARMACOLOGY API. | |
| Based on the user's natural language request, determine the appropriate GtoPdb API endpoint and parameters. | |
| GTOPDB API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The full API endpoint to query | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - Main endpoints include: "targets", "ligands", "interactions", "diseases", "refs" | |
| - Target types include: "GPCR", "NHR", "LGIC", "VGIC", "OtherIC", "Enzyme", "CatalyticReceptor", "Transporter", "OtherProtein" | |
| - Ligand types include: "Synthetic organic", "Metabolite", "Natural product", "Endogenous peptide", "Peptide", "Antibody", "Inorganic", "Approved", "Withdrawn", "Labelled", "INN" | |
| - Interaction types include: "Activator", "Agonist", "Allosteric modulator", "Antagonist", "Antibody", "Channel blocker", "Gating inhibitor", "Inhibitor", "Subunit-specific" | |
| - For specific target/ligand details, use formats like "targets/{{targetId}}" or "ligands/{{ligandId}}" | |
| - For subresources, use formats like "targets/{{targetId}}/interactions" or "ligands/{{ligandId}}/structure" | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=gtopdb_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| description = f"Direct query to GtoPdb {endpoint}" | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = f"Direct query to provided endpoint" | |
| # Execute the GtoPdb API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| # Format the results if not verbose and successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| api_result["result"] = _format_query_results(api_result["result"]) | |
| return api_result | |
| def region_to_ccre_screen(coord_chrom: str, coord_start: int, coord_end: int, assembly: str = "GRCh38" ) -> str: | |
| """ | |
| Given starting and ending coordinates, this function retrieves information of intersecting cCREs. | |
| Args: | |
| coord_chrom: Chromosome of the gene, formatted like 'chr12' | |
| coord_start: Starting chromosome coordinate | |
| coord_end: Ending chromosome coordinate | |
| assembly: Assembly of the genome, formatted like 'GRCh38'. Default is 'GRCh38' | |
| Returns: | |
| A detailed string explaining the steps and the intersecting cCRE data or any error encountered | |
| """ | |
| steps = [] | |
| try: | |
| steps.append(f"Starting cCRE data retrieval for coordinates: {coord_chrom}:{coord_start}-{coord_end} (Assembly: {assembly}).") | |
| # Build the URL and request payload | |
| url = "https://screen-beta-api.wenglab.org/dataws/cre_table" | |
| data = { | |
| "assembly": assembly, | |
| "coord_chrom": coord_chrom, | |
| "coord_start": coord_start, | |
| "coord_end": coord_end | |
| } | |
| steps.append("Sending POST request to API with the following data:") | |
| steps.append(str(data)) | |
| # Make the request | |
| response = requests.post(url, json=data) | |
| # Check if the response is successful | |
| if not response.ok: | |
| raise Exception(f"Request failed with status code {response.status_code}. Response: {response.text}") | |
| steps.append("Request executed successfully. Parsing the response...") | |
| # Parse the JSON response | |
| response_json = response.json() | |
| if "errors" in response_json: | |
| raise Exception(f"API error: {response_json['errors']}") | |
| # Function to reduce and filter response data | |
| def reduce_tokens(res_json): | |
| # Remove unnecessary fields and round floats | |
| res = sorted(res_json["cres"], key=lambda x: x['dnase_zscore'], reverse=True) | |
| filtered_res = [] | |
| for item in res: | |
| new_item = { | |
| 'chrom': item['chrom'], | |
| 'start': item['start'], | |
| 'len': item['len'], | |
| 'pct': item['pct'], | |
| 'ctcf_zscore': round(item['ctcf_zscore'], 2), | |
| 'dnase_zscore': round(item['dnase_zscore'], 2), | |
| 'enhancer_zscore': round(item['enhancer_zscore'], 2), | |
| 'promoter_zscore': round(item['promoter_zscore'], 2), | |
| 'accession': item['info']['accession'], | |
| 'isproximal': item['info']['isproximal'], | |
| 'concordance': item['info']['concordant'], | |
| 'ctcfmax': round(item['info']['ctcfmax'], 2), | |
| 'k4me3max': round(item['info']['k4me3max'], 2), | |
| 'k27acmax': round(item['info']['k27acmax'], 2) | |
| } | |
| filtered_res.append(new_item) | |
| return filtered_res | |
| # Process the response data | |
| filtered_data = reduce_tokens(response_json) | |
| if not filtered_data: | |
| steps.append(f"No intersecting cCREs found for coordinates: {coord_chrom}:{coord_start}-{coord_end}.") | |
| return "\n".join(steps + ["No cCRE data available for this genomic region."]) | |
| # Format the result into a readable string | |
| ccre_data_string = f"Intersecting cCREs for {coord_chrom}:{coord_start}-{coord_end} (Assembly: {assembly}):\n" | |
| for i, ccre in enumerate(filtered_data, 1): | |
| ccre_data_string += ( | |
| f"cCRE {i}:\n" | |
| f" Chromosome: {ccre['chrom']}\n" | |
| f" Start: {ccre['start']}\n" | |
| f" Length: {ccre['len']}\n" | |
| f" PCT: {ccre['pct']}\n" | |
| f" CTCF Z-score: {ccre['ctcf_zscore']}\n" | |
| f" DNase Z-score: {ccre['dnase_zscore']}\n" | |
| f" Enhancer Z-score: {ccre['enhancer_zscore']}\n" | |
| f" Promoter Z-score: {ccre['promoter_zscore']}\n" | |
| f" Accession: {ccre['accession']}\n" | |
| f" Is Proximal: {ccre['isproximal']}\n" | |
| f" Concordance: {ccre['concordance']}\n" | |
| f" CTCFmax: {ccre['ctcfmax']}\n" | |
| f" K4me3max: {ccre['k4me3max']}\n" | |
| f" K27acmax: {ccre['k27acmax']}\n\n" | |
| ) | |
| steps.append(f"cCRE data successfully retrieved and formatted for {coord_chrom}:{coord_start}-{coord_end}.") | |
| return "\n".join(steps + [ccre_data_string]) | |
| except Exception as e: | |
| steps.append(f"Exception encountered: {str(e)}") | |
| return "\n".join(steps + [f"Error: {str(e)}"]) | |
| def get_genes_near_ccre(accession: str, assembly: str, chromosome: str, k: int = 10) -> str: | |
| """ | |
| Given a cCRE (Candidate cis-Regulatory Element), this function returns a string containing the | |
| steps it performs and the k nearest genes sorted by distance. | |
| Args: | |
| accession: ENCODE Accession ID of query cCRE, e.g., EH38E1516980 | |
| assembly: Assembly of the gene, e.g., 'GRCh38' | |
| chromosome: Chromosome of the gene, e.g., 'chr12' | |
| k: Number of nearby genes to return, sorted by distance. Default is 10 | |
| Returns: | |
| Steps performed and the result | |
| """ | |
| steps_log = f"Starting process with accession: {accession}, assembly: {assembly}, chromosome: {chromosome}, k: {k}\n" | |
| url = "https://screen-beta-api.wenglab.org/dataws/re_detail/nearbyGenomic" | |
| data = { | |
| "accession": accession, | |
| "assembly": assembly, | |
| "coord_chrom": chromosome | |
| } | |
| steps_log += "Sending POST request to API with given data.\n" | |
| response = requests.post(url, json=data) | |
| if not response.ok: | |
| steps_log += f"API request failed with response: {response.text}\n" | |
| return steps_log | |
| response_json = response.json() | |
| if "errors" in response_json: | |
| steps_log += f"API returned errors: {response_json['errors']}\n" | |
| return steps_log | |
| nearby_genes = response_json.get(accession, {}).get("nearby_genes", []) | |
| if not nearby_genes: | |
| steps_log += "No nearby genes found for the given accession.\n" | |
| return steps_log | |
| steps_log += "Successfully retrieved nearby genes. Sorting them by distance.\n" | |
| sorted_genes = sorted(nearby_genes, key=lambda x: x['distance'])[:k] | |
| steps_log += f"Returning the top {k} nearest genes.\n" | |
| steps_log += "Result:\n" | |
| for gene in sorted_genes: | |
| gene_name = gene.get('name', 'Unknown') | |
| distance = gene.get('distance', 'N/A') | |
| ensembl_id = gene.get('ensemblid_ver', 'N/A') | |
| start = gene.get('start', 'N/A') | |
| stop = gene.get('stop', 'N/A') | |
| chrom = gene.get('chrom', 'N/A') | |
| steps_log += f"Gene: {gene_name}, Distance: {distance}, Ensembl ID: {ensembl_id}, Chromosome: {chrom}, Start: {start}, Stop: {stop}\n" | |
| return steps_log | |
| def query_remap(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the ReMap database for regulatory elements and transcription factor binding sites. | |
| Args: | |
| prompt (str, required): Natural language query about transcription factors and binding sites | |
| endpoint (str, optional): Full API endpoint to query (e.g., "https://remap.univ-amu.fr/api/v1/catalogue/tf?tf=CTCF") | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_remap("Find CTCF binding sites in chromosome 1") | |
| - Direct endpoint: query_remap(endpoint="catalogue/tf", params={"tf": "CTCF"}) | |
| """ | |
| # Base URL for ReMap API | |
| base_url = "https://remap.univ-amu.fr/api/v1" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load ReMap schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "remap.pkl") | |
| with open(schema_path, "rb") as f: | |
| remap_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a genomics expert specialized in using the ReMap database API. | |
| Based on the user's natural language request, determine the appropriate ReMap API endpoint and parameters. | |
| REMAP API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The full url endpoint to query | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - ReMap is a database of regulatory regions and transcription factor binding sites based on ChIP-seq experiments | |
| - Common endpoints include: "catalogue/tf" (transcription factors), "catalogue/biotype" (biotypes), "browse/peaks" (binding sites) | |
| - For searching binding sites, you can filter by transcription factor (tf), cell line, biotype, chromosome, etc. | |
| - Genomic coordinates should be specified with "chr", "start", and "end" parameters | |
| - For limiting results, use "limit" parameter (default is 100) | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=remap_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| description = f"Direct query to ReMap {endpoint}" | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = f"Direct query to provided endpoint" | |
| # Execute the ReMap API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| # Format the results if not verbose and successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| api_result["result"] = _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_mpd(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the Mouse Phenome Database (MPD) for mouse strain phenotype data. | |
| Args: | |
| prompt (str, required): Natural language query about mouse phenotypes, strains, or measurements | |
| endpoint (str, optional): Full API endpoint to query (e.g., "https://phenomedoc.jax.org/MPD_API/strains") | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_mpd("Find phenotype data for C57BL/6J mice related to blood glucose") | |
| - Direct endpoint: query_mpd(endpoint="strains/C57BL/6J/measures") | |
| """ | |
| # Base URL for MPD API | |
| base_url = "https://phenome.jax.org" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load MPD schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "mpd.pkl") | |
| with open(schema_path, "rb") as f: | |
| mpd_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a mouse genetics expert specialized in using the Mouse Phenome Database (MPD) API. | |
| Based on the user's natural language request, determine the appropriate MPD API endpoint and parameters. | |
| MPD API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The full url endpoint to query (e.g. https://phenome.jax.org/api/strains) | |
| 2. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - The MPD contains phenotype data for diverse strains of laboratory mice | |
| - Common endpoints include: "strains" (mouse strains), "measures" (phenotypic measurements), "genes" (gene info) | |
| - Use the url to construct the endpoint, not the endpoint name | |
| - Common mouse strains include: "C57BL/6J", "DBA/2J", "BALB/cJ", "A/J", "129S1/SvImJ" | |
| - Common phenotypic domains include: "behavior", "blood_chemistry", "body_weight", "cardiovascular", "growth", "metabolism" | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=mpd_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| description = f"Direct query to MPD {endpoint}" | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = f"Direct query to provided endpoint" | |
| # Execute the MPD API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| description=description | |
| ) | |
| # Format the results if not verbose and successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| api_result["result"] = _format_query_results(api_result["result"]) | |
| return api_result | |
| def query_emdb(prompt: str = None, endpoint: str = None, api_key: str = None, verbose: bool = True) -> dict: | |
| """ | |
| Query the Electron Microscopy Data Bank (EMDB) for 3D macromolecular structures. | |
| Args: | |
| prompt (str, required): Natural language query about EM structures and associated data | |
| endpoint (str, optional): Full API endpoint to query (e.g., "https://www.ebi.ac.uk/emdb/api/search") | |
| api_key (str, optional): Anthropic API key. If None, will use ANTHROPIC_API_KEY env variable | |
| verbose (bool): Whether to return detailed results | |
| verbose (bool): Whether to return detailed results | |
| Returns: | |
| dict: Dictionary containing the query results or error information | |
| Examples: | |
| - Natural language: query_emdb("Find cryo-EM structures of ribosomes at resolution better than 3Å") | |
| - Direct endpoint: query_emdb(endpoint="entry/EMD-10000") | |
| """ | |
| # Base URL for EMDB API | |
| base_url = "https://www.ebi.ac.uk/emdb/api" | |
| # Ensure we have either a prompt or an endpoint | |
| if prompt is None and endpoint is None: | |
| return {"error": "Either a prompt or an endpoint must be provided"} | |
| # If using prompt, parse with Claude | |
| if prompt: | |
| # Load EMDB schema | |
| schema_path = os.path.join(SCHEMA_DB_PATH, "emdb.pkl") | |
| with open(schema_path, "rb") as f: | |
| emdb_schema = pickle.load(f) | |
| # Create system prompt template | |
| system_template = """ | |
| You are a structural biology expert specialized in using the Electron Microscopy Data Bank (EMDB) API. | |
| Based on the user's natural language request, determine the appropriate EMDB API endpoint and parameters. | |
| EMDB API SCHEMA: | |
| {schema} | |
| Your response should be a JSON object with the following fields: | |
| 1. "endpoint": The API endpoint to query (e.g., "search", "entry/EMD-XXXXX") | |
| 2. "params": An object containing query parameters specific to the endpoint | |
| 3. "description": A brief description of what the query is doing | |
| SPECIAL NOTES: | |
| - EMDB contains 3D macromolecular structures determined by electron microscopy | |
| - Common endpoints include: "search" (search for entries), "entry/EMD-XXXXX" (specific entry details) | |
| - For searching, you can filter by resolution, specimen, authors, release date, etc. | |
| - Resolution filters should be specified with "resolution_low" and "resolution_high" parameters | |
| - For specific entry retrieval, use the format "entry/EMD-XXXXX" where XXXXX is the EMDB ID | |
| - Common specimen types include: "ribosome", "virus", "membrane protein", "filament" | |
| Return ONLY the JSON object with no additional text. | |
| """ | |
| # Query Gemini to generate the API call | |
| gemini_result = _query_gemini_for_api( | |
| prompt=prompt, | |
| schema=emdb_schema, | |
| system_template=system_template | |
| ) | |
| if not gemini_result["success"]: | |
| return gemini_result | |
| # Get the endpoint and parameters from Gemini's response | |
| query_info = gemini_result["data"] | |
| endpoint = query_info.get("endpoint", "") | |
| params = query_info.get("params", {}) | |
| description = query_info.get("description", "") | |
| if not endpoint: | |
| return { | |
| "error": "Failed to generate a valid endpoint from the prompt", | |
| "gemini_response": gemini_result.get("raw_response", "No response") | |
| } | |
| else: | |
| # Process provided endpoint | |
| params = {} | |
| description = f"Direct query to EMDB {endpoint}" | |
| # Remove leading slash if present | |
| if endpoint.startswith("/"): | |
| endpoint = f"{base_url}{endpoint}" | |
| elif not endpoint.startswith("http"): | |
| endpoint = f"{base_url}/{endpoint.lstrip('/')}" | |
| description = f"Direct query to provided endpoint" | |
| # Execute the EMDB API request using the helper function | |
| api_result = _query_rest_api( | |
| endpoint=endpoint, | |
| method="GET", | |
| params=params, | |
| description=description | |
| ) | |
| # Format the results if not verbose and successful | |
| if not verbose and "success" in api_result and api_result["success"] and "result" in api_result: | |
| api_result["result"] = _format_query_results(api_result["result"]) | |
| return api_result | |