BOTeome / scripts /string_db.py
juan
string lookup api
514d14b
Raw
History Blame Contribute Delete
7.92 kB
import time
import requests
_API = 'https://string-db.org/api'
_TIMEOUT = 60
_CALLER = 'BOTeome'
# STRING asks callers to stay under roughly one request per second.
_MIN_INTERVAL = 1.0
_last_request = 0.0
MOUSE = 10090
HUMAN = 9606
# Enrichment returns a long tail of publication (PMID) hits that crowd out the
# interpretable annotations, so keep a curated set unless asked for everything.
_DEFAULT_CATEGORIES = (
'Process', 'Function', 'Component', 'KEGG', 'RCTM', 'WikiPathways',
'Pfam', 'InterPro', 'SMART', 'Keyword', 'DISEASES',
)
_CATEGORY_LABELS = {
'Process': 'GO biological process',
'Function': 'GO molecular function',
'Component': 'GO cellular component',
'KEGG': 'KEGG pathway',
'RCTM': 'Reactome pathway',
'WikiPathways': 'WikiPathways',
'Pfam': 'Pfam domain',
'InterPro': 'InterPro domain',
'SMART': 'SMART domain',
'Keyword': 'UniProt keyword',
'DISEASES': 'Disease association',
'COMPARTMENTS': 'Subcellular compartment',
'TISSUES': 'Tissue expression',
'PMID': 'Publication',
}
def _get(method, **params):
"""Throttled JSON request against the STRING API."""
global _last_request
wait = _MIN_INTERVAL - (time.monotonic() - _last_request)
if wait > 0:
time.sleep(wait)
params['caller_identity'] = _CALLER
r = requests.get(f'{_API}/json/{method}', params=params, timeout=_TIMEOUT)
_last_request = time.monotonic()
if r.status_code == 400:
# STRING explains bad species / unknown identifiers in the body.
raise ValueError(f'STRING rejected the request: {r.text.strip()[:300]}')
if r.status_code == 404:
# Returned when no identifier maps to a STRING protein.
raise ValueError(
'STRING recognised none of those identifiers. Check the species is '
'right, and prefer gene symbols (Gfap) over UniProt accessions.'
)
r.raise_for_status()
return r.json()
def _join(identifiers):
"""STRING separates identifiers with a carriage return (the %0d in its docs).
Pass the raw character: requests percent-encodes the query string itself.
"""
if isinstance(identifiers, str):
identifiers = [i.strip() for i in identifiers.replace(',', '\n').splitlines()]
return '\r'.join(i for i in (str(x).strip() for x in identifiers) if i)
def map_identifiers(identifiers, species=MOUSE):
"""Resolve gene names or UniProt accessions to STRING proteins."""
return _get(
'get_string_ids',
identifiers=_join(identifiers),
species=species,
limit=1,
echo_query=1,
)
def interaction_partners(identifiers, species=MOUSE, limit=10, required_score=400):
"""Highest-confidence known partners of the given proteins."""
return _get(
'interaction_partners',
identifiers=_join(identifiers),
species=species,
limit=limit,
required_score=required_score,
)
def network_interactions(identifiers, species=MOUSE, required_score=400):
"""Interactions *among* the given proteins, ignoring outside partners."""
return _get(
'network',
identifiers=_join(identifiers),
species=species,
required_score=required_score,
)
def functional_enrichment(identifiers, species=MOUSE):
"""GO / pathway / domain enrichment for a protein set."""
return _get('enrichment', identifiers=_join(identifiers), species=species)
# --- Formatting ------------------------------------------------------
def format_mapping(rows):
if not rows:
return 'None of those identifiers matched a STRING protein.'
lines = []
for r in rows:
annotation = (r.get('annotation') or '').strip()
if len(annotation) > 300:
annotation = annotation[:300].rstrip() + '...'
lines.append(
f"{r.get('queryItem', '?')} -> {r.get('preferredName')} "
f"({r.get('stringId')}, {r.get('taxonName')})\n {annotation}"
)
return '\n'.join(lines)
def format_partners(rows):
if not rows:
return ('No interaction partners above the confidence threshold. '
'Try a lower required_score.')
lines = ['Interaction partners (STRING combined score, 0-1):']
for r in sorted(rows, key=lambda x: x.get('score', 0), reverse=True):
lines.append(
f" {r.get('preferredName_A')} -- {r.get('preferredName_B')} "
f"score={r.get('score')} "
f"(experimental={r.get('escore')}, database={r.get('dscore')}, "
f"coexpression={r.get('ascore')}, textmining={r.get('tscore')})"
)
return '\n'.join(lines)
def format_network(rows, queried=None):
if not rows:
return 'No interactions found among those proteins at this confidence level.'
lines = [f'{len(rows)} interactions among the queried proteins:']
for r in sorted(rows, key=lambda x: x.get('score', 0), reverse=True):
lines.append(
f" {r.get('preferredName_A')} -- {r.get('preferredName_B')} "
f"score={r.get('score')}"
)
if queried:
connected = {r.get('preferredName_A') for r in rows} | {r.get('preferredName_B') for r in rows}
lines.append(f'\n{len(connected)} of the queried proteins appear in this network.')
return '\n'.join(lines)
def format_enrichment(rows, categories=None, top_per_category=10, fdr_cutoff=0.05):
if not rows:
return 'No functional enrichment found for that protein set.'
allowed = None if categories == 'all' else set(categories or _DEFAULT_CATEGORIES)
kept = [
r for r in rows
if (allowed is None or r.get('category') in allowed)
and r.get('fdr', 1) <= fdr_cutoff
]
if not kept:
return (f'No terms passed FDR <= {fdr_cutoff} in the selected categories '
f'(STRING returned {len(rows)} raw terms).')
by_category = {}
for r in kept:
by_category.setdefault(r['category'], []).append(r)
blocks = [f'Functional enrichment ({len(kept)} terms at FDR <= {fdr_cutoff}):']
for category, terms in sorted(by_category.items(), key=lambda kv: -len(kv[1])):
label = _CATEGORY_LABELS.get(category, category)
terms.sort(key=lambda x: x.get('fdr', 1))
lines = [f'\n## {label}']
for t in terms[:top_per_category]:
genes = ', '.join(t.get('preferredNames', []))
lines.append(
f" {t.get('description')} [{t.get('term')}]\n"
f" FDR={t.get('fdr'):.2g} "
f"{t.get('number_of_genes')}/{t.get('number_of_genes_in_background')} genes: {genes}"
)
if len(terms) > top_per_category:
lines.append(f' ... and {len(terms) - top_per_category} more {label} terms')
blocks.append('\n'.join(lines))
return '\n'.join(blocks)
# --- Error-tolerant entry points -------------------------------------
def lookup_proteins(identifiers, species=MOUSE):
try:
return format_mapping(map_identifiers(identifiers, species))
except Exception as e:
return f'STRING lookup failed: {e}'
def get_partners(identifiers, species=MOUSE, limit=10, required_score=400):
try:
return format_partners(interaction_partners(identifiers, species, limit, required_score))
except Exception as e:
return f'STRING interaction lookup failed: {e}'
def get_network(identifiers, species=MOUSE, required_score=400):
try:
return format_network(network_interactions(identifiers, species, required_score), identifiers)
except Exception as e:
return f'STRING network lookup failed: {e}'
def get_enrichment(identifiers, species=MOUSE, categories=None):
try:
return format_enrichment(functional_enrichment(identifiers, species), categories)
except Exception as e:
return f'STRING enrichment failed: {e}'