Spaces:
Sleeping
Sleeping
| """ | |
| AlphaFold Database and UniProt API clients. | |
| All data fetched live from official public APIs. | |
| """ | |
| from __future__ import annotations | |
| import httpx | |
| from typing import Any | |
| ALPHAFOLD_BASE = "https://alphafold.ebi.ac.uk/api" | |
| UNIPROT_BASE = "https://rest.uniprot.org/uniprotkb" | |
| TIMEOUT = 20.0 | |
| async def fetch_alphafold_summary(uniprot_id: str) -> dict[str, Any]: | |
| """Fetch AlphaFold structure summary for a UniProt ID.""" | |
| uid = uniprot_id.strip().upper() | |
| url = f"{ALPHAFOLD_BASE}/prediction/{uid}" | |
| async with httpx.AsyncClient(timeout=TIMEOUT) as client: | |
| r = await client.get(url) | |
| r.raise_for_status() | |
| data = r.json() | |
| if isinstance(data, list) and data: | |
| return data[0] | |
| return data | |
| async def fetch_uniprot_entry(uniprot_id: str) -> dict[str, Any]: | |
| """Fetch UniProt entry metadata.""" | |
| uid = uniprot_id.strip().upper() | |
| url = f"{UNIPROT_BASE}/{uid}.json" | |
| async with httpx.AsyncClient(timeout=TIMEOUT) as client: | |
| r = await client.get(url) | |
| r.raise_for_status() | |
| return r.json() | |
| async def fetch_pae_image_url(uniprot_id: str) -> str | None: | |
| """Return URL of the predicted aligned error (PAE) image.""" | |
| try: | |
| summary = await fetch_alphafold_summary(uniprot_id) | |
| return summary.get("paeImageUrl") or summary.get("pae_image_url") | |
| except Exception: | |
| return None | |
| def extract_protein_metadata(uniprot_data: dict) -> dict[str, Any]: | |
| """Parse UniProt JSON into clean metadata dict.""" | |
| entry = uniprot_data | |
| # Protein name | |
| pn = entry.get("proteinDescription", {}) | |
| recommended = pn.get("recommendedName", {}) | |
| full_name = ( | |
| recommended.get("fullName", {}).get("value", "") | |
| or pn.get("submissionNames", [{}])[0].get("fullName", {}).get("value", "") | |
| ) | |
| # Gene names | |
| genes = [g.get("geneName", {}).get("value", "") for g in entry.get("genes", [])] | |
| gene_str = ", ".join(g for g in genes if g) | |
| # Organism | |
| organism = entry.get("organism", {}).get("scientificName", "") | |
| # Sequence | |
| seq_data = entry.get("sequence", {}) | |
| sequence = seq_data.get("value", "") | |
| length = seq_data.get("length", len(sequence)) | |
| # Function annotation | |
| comments = entry.get("comments", []) | |
| function_text = "" | |
| for c in comments: | |
| if c.get("commentType") == "FUNCTION": | |
| texts = c.get("texts", []) | |
| if texts: | |
| function_text = texts[0].get("value", "")[:400] | |
| break | |
| # Keywords for domain classification | |
| keywords = [kw.get("name", "") for kw in entry.get("keywords", [])] | |
| # Subcellular location | |
| location = "" | |
| for c in comments: | |
| if c.get("commentType") == "SUBCELLULAR LOCATION": | |
| locs = c.get("subcellularLocations", []) | |
| if locs: | |
| location = locs[0].get("location", {}).get("value", "") | |
| break | |
| # Disease associations | |
| diseases = [] | |
| for c in comments: | |
| if c.get("commentType") == "DISEASE": | |
| d = c.get("disease", {}) | |
| if d.get("diseaseId"): | |
| diseases.append(d["diseaseId"]) | |
| # Infer functional category from keywords | |
| kw_lower = " ".join(keywords).lower() | |
| if any(t in kw_lower for t in ["transferase", "hydrolase", "lyase", "kinase", "enzyme"]): | |
| function_category = "enzyme" | |
| elif any(t in kw_lower for t in ["receptor", "ligand-bind"]): | |
| function_category = "receptor" | |
| elif any(t in kw_lower for t in ["transport", "channel", "carrier"]): | |
| function_category = "transporter" | |
| elif any(t in kw_lower for t in ["structural", "cytoskeleton", "collagen"]): | |
| function_category = "structural" | |
| elif any(t in kw_lower for t in ["transcription", "dna-binding", "chromatin"]): | |
| function_category = "transcription" | |
| elif any(t in kw_lower for t in ["signal", "hormone", "cytokine"]): | |
| function_category = "signaling" | |
| else: | |
| function_category = "unknown" | |
| return { | |
| "uniprot_id": entry.get("primaryAccession", ""), | |
| "protein_name": full_name, | |
| "gene_name": gene_str, | |
| "organism": organism, | |
| "sequence": sequence, | |
| "length": length, | |
| "function_text": function_text, | |
| "function_category": function_category, | |
| "keywords": keywords[:12], | |
| "subcellular_location": location, | |
| "disease_associations": diseases[:6], | |
| } | |
| async def fetch_plddt_scores(alphafold_data: dict) -> list[float]: | |
| """ | |
| Fetch per-residue pLDDT scores from AlphaFold confidence JSON file. | |
| The summary endpoint doesn't include scores inline — they're in a separate file. | |
| Falls back gracefully if unavailable. | |
| """ | |
| # Try the confidence JSON URL directly | |
| conf_url = alphafold_data.get("confidenceUrl") or alphafold_data.get("confidence_url") | |
| if not conf_url: | |
| # Construct from entry ID and version | |
| entry_id = alphafold_data.get("entryId", "") | |
| version = alphafold_data.get("latestVersion", 4) | |
| if entry_id: | |
| conf_url = f"https://alphafold.ebi.ac.uk/files/{entry_id}-confidence_v{version}.json" | |
| if conf_url: | |
| try: | |
| async with httpx.AsyncClient(timeout=TIMEOUT) as client: | |
| r = await client.get(conf_url) | |
| r.raise_for_status() | |
| data = r.json() | |
| scores = data.get("confidenceScore") or data if isinstance(data, list) else [] | |
| if scores: | |
| return [float(s) for s in scores] | |
| except Exception: | |
| pass | |
| # Final fallback: single average value from summary | |
| avg = (alphafold_data.get("confidenceAvgLocalScore") or | |
| alphafold_data.get("plddt") or 75.0) | |
| return [float(avg)] | |
| def extract_plddt_scores(alphafold_data: dict) -> list[float]: | |
| """Synchronous fallback — use fetch_plddt_scores() for accurate per-residue data.""" | |
| avg = (alphafold_data.get("confidenceAvgLocalScore") or | |
| alphafold_data.get("plddt") or 75.0) | |
| return [float(avg)] | |