lvwerra's picture
lvwerra HF Staff
Keep the match list and the chart in step (#10)
a537c09
Raw History Blame Contribute Delete
33.5 kB
"""Read-only taxonomy navigation, independent of the annotation lookup index."""
from collections import Counter
from contextlib import closing
import math
from html import escape
import json
from pathlib import Path
import sqlite3
import gradio as gr
ROOT = Path(__file__).resolve().parent
DATABASE = ROOT / "data/taxonomy.sqlite"
COMMON_NAMES = ROOT / "data/common_names.json"
ICONS = ROOT / "data/icons.json"
ICON_DIR = ROOT / "data/icons"
# Gradio serves whitelisted files from this route; see build_taxonomy_tab.
ICON_ROUTE = "/gradio_api/file="
LICENSE_NAMES = {"https://creativecommons.org/publicdomain/zero/1.0/": "CC0 1.0",
"https://creativecommons.org/licenses/by/4.0/": "CC BY 4.0",
"https://creativecommons.org/licenses/by/3.0/": "CC BY 3.0",
"https://creativecommons.org/publicdomain/mark/1.0/": "Public Domain Mark 1.0"}
EUKARYOTA = 2759
EUK_TREE = """WITH RECURSIVE euk(taxid) AS (
SELECT 2759 UNION ALL
SELECT t.taxid FROM taxa t JOIN euk e ON t.parent_id=e.taxid
WHERE t.taxid != t.parent_id
) """
# Scientific names stay the primary label; English glosses are a secondary line.
# See refresh_common_names.py for how the lookup is built.
MAX_ENGLISH = 28
MAX_LABEL = 22
# Jumping-off points above the tree. Each one expands its whole lineage, so a
# deep pick like humans is also the quickest way to see what a full path looks
# like. Order runs broad to narrow.
SHORTCUTS = ((33208, "Animals"), (40674, "Mammals"), (9606, "Humans"), (8782, "Birds"),
(7898, "Ray-finned fish"), (50557, "Insects"), (6142, "Jellyfish"),
(33090, "Green plants"), (4751, "Fungi"))
COLUMN_LIMIT = 6
SEARCH_LIMIT = 12
MAX_DEPTH = 80
DEFAULT_PATH = "2759/33154"
def encode_path(steps):
"""Serialize selected (taxid, offset) pairs; an offset marks an expanded "Other lineages" aggregate."""
return "/".join(f"{t}~{o}" if o else str(t) for t, o in steps)
class Taxonomy:
def __init__(self, path=DATABASE, common_names=COMMON_NAMES, icons=ICONS):
self.path = Path(path)
self.english, self.english_sources = {}, {}
if Path(common_names).exists():
payload = json.loads(Path(common_names).read_text())
self.english = {int(t): name for t, name in payload["names"].items()}
self.english_sources = payload.get("sources", {})
self.icons, self.icon_credits, self.icon_meta = {}, {}, {}
if Path(icons).exists():
payload = json.loads(Path(icons).read_text())
self.icon_meta = {k: v for k, v in payload.items() if k not in ("images", "taxa")}
self.icon_credits = payload.get("images", {})
self.icons = {int(t): image for t, image in payload.get("taxa", {}).items()
if (ICON_DIR / f"{image}.svg").exists()}
self._inherited_icons = {}
with closing(self.connect()) as conn:
self.metadata = json.loads(conn.execute("SELECT value FROM metadata").fetchone()[0])
self.eukaryote_ids = frozenset(r[0] for r in conn.execute(EUK_TREE + "SELECT taxid FROM euk"))
self.eukaryotes = dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (EUKARYOTA,)).fetchone())
self.coverage = self.metadata.get("coverage")
def connect(self):
conn = sqlite3.connect(f"{self.path.resolve().as_uri()}?mode=ro", uri=True)
conn.row_factory = sqlite3.Row
return conn
def closeness(self, row, needle):
"""Whole-name matches first, then word starts: 'thale' should find thale cress
before it finds Magnaporthales."""
names = [row['name'].casefold(), self.english.get(row['taxid'], "").casefold()]
if any(name == needle for name in names):
return 0
if any(name.startswith(needle) for name in names):
return 1
if any(word.startswith(needle) for name in names for word in name.split()):
return 2
return 3
def label(self, row):
english = self.english.get(row['taxid'], "")
return (f"{row['name']}" + (f" · {english}" if english else "") +
f" · {row['total_count']:,} assemblies · taxid {row['taxid']}")
def search(self, query, limit=SEARCH_LIMIT):
"""Match a taxon id, a scientific name, or one of the English names.
The eukaryote set is already in memory, so the filtering happens here
rather than in a recursive CTE the query would otherwise re-walk on every
keystroke.
"""
query = str(query or "").strip()
if len(query) < 2 and not query.isdigit():
return []
needle = query.casefold()
with closing(self.connect()) as conn:
if query.lstrip("-").isdigit():
row = conn.execute("SELECT * FROM taxa WHERE taxid=?", (int(query),)).fetchone()
rows = [row] if row and row['taxid'] in self.eukaryote_ids else []
else:
pattern = "%" + query.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_") + "%"
found = {r['taxid']: r for r in conn.execute(
"SELECT * FROM taxa WHERE name LIKE ? ESCAPE '\\' ORDER BY total_count DESC LIMIT ?",
(pattern, limit * 4)) if r['taxid'] in self.eukaryote_ids}
english = [t for t, name in self.english.items()
if needle in name.casefold() and t not in found and t in self.eukaryote_ids]
for chunk in (english[i:i + 400] for i in range(0, len(english), 400)):
marks = ",".join("?" * len(chunk))
found.update({r['taxid']: r for r in conn.execute(
f"SELECT * FROM taxa WHERE taxid IN ({marks})", chunk)})
rows = sorted(found.values(), key=lambda r: (self.closeness(r, needle), -r['total_count']))[:limit]
return [(self.label(r), str(r['taxid'])) for r in rows]
def icon_for(self, conn, taxid):
"""Nearest icon at or above this taxon, following the lineage to Eukaryota.
Returns the image and whether it belongs to this taxon or was borrowed.
"""
if taxid in self.icons:
return self.icons[taxid], False
if taxid in self._inherited_icons:
return self._inherited_icons[taxid], True
walked, current = [], taxid
for _ in range(MAX_DEPTH):
row = conn.execute("SELECT parent_id FROM taxa WHERE taxid=?", (current,)).fetchone()
if row is None or row['parent_id'] == current:
break
walked.append(current)
current = row['parent_id']
if current in self.icons or current in self._inherited_icons:
break
found = self.icons.get(current) or self._inherited_icons.get(current, "")
for step in walked:
self._inherited_icons[step] = found
return found, True
def lineage(self, conn, taxid):
rows = [dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (taxid,)).fetchone())]
while rows[-1]['taxid'] != EUKARYOTA:
rows.append(dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (rows[-1]['parent_id'],)).fetchone()))
return rows[::-1]
def assemblies_for(self, path, limit=1):
"""Annotated accessions under the taxon a route selects, largest group first.
The recursive walk starts at the selected taxon rather than at Eukaryota,
so it touches only that subtree.
"""
try:
taxid = int(str(path).split("/")[-1].partition("~")[0])
except (ValueError, AttributeError):
return []
if taxid not in self.eukaryote_ids:
return []
with closing(self.connect()) as conn:
if not conn.execute("SELECT name FROM sqlite_master WHERE type='table' AND name='assemblies'").fetchone():
return []
return [r[0] for r in conn.execute(
"""WITH RECURSIVE sub(taxid) AS (
SELECT ? UNION ALL
SELECT t.taxid FROM taxa t JOIN sub ON t.parent_id=sub.taxid WHERE t.taxid!=t.parent_id
) SELECT accession FROM assemblies JOIN sub USING(taxid) ORDER BY accession LIMIT ?""",
(taxid, limit))]
def selected_name(self, path):
taxid = int(str(path).split("/")[-1].partition("~")[0])
with closing(self.connect()) as conn:
row = conn.execute("SELECT name FROM taxa WHERE taxid=?", (taxid,)).fetchone()
return row['name'] if row else str(taxid)
def taxon_path(self, taxid):
"""Path that expands every ancestor of a taxon, for search results and shortcuts."""
taxid = int(taxid)
if taxid not in self.eukaryote_ids:
raise ValueError("Choose a eukaryotic taxon from this snapshot.")
with closing(self.connect()) as conn:
return encode_path((r['taxid'], 0) for r in self.lineage(conn, taxid))
def view_taxon(self, taxid):
return self.view(self.taxon_path(taxid))
def view(self, path=DEFAULT_PATH):
"""Render one column per expanded group. A path lists the selected node of each column."""
try:
steps = [(int(t), int(o or 0)) for t, _, o in (part.partition("~") for part in str(path).split("/"))]
except ValueError as exc:
raise ValueError("Choose a eukaryotic taxon from this snapshot.") from exc
if steps[0] != (EUKARYOTA, 0) or len(steps) > MAX_DEPTH or any(
t not in self.eukaryote_ids or o < 0 for t, o in steps):
raise ValueError("Choose a eukaryotic taxon from this snapshot.")
with closing(self.connect()) as conn:
def taxon(t):
return dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (t,)).fetchone())
def node(row, offset=0, **extra):
return dict(taxid=row['taxid'], offset=offset, name=row['name'], label=row['name'],
english=self.english.get(row['taxid'], ""),
**dict(zip(("icon", "borrowed_icon"), self.icon_for(conn, row['taxid']))),
total=row['total_count'], covered=row.get('covered_total', 0), **extra)
def remainder(rows, parent, start):
return dict(taxid=parent, offset=start, name=f"Other lineages · {len(rows):,} groups (display aggregate)",
label=f"Other lineages ({len(rows):,})", english="", icon="", borrowed_icon=False, total=sum(r['total_count'] for r in rows),
covered=sum(r.get('covered_total', 0) for r in rows), aggregate=True)
def column(step, selected):
"""Direct lineages of a selected node, keeping the next selection visible."""
parent, offset = step
rows = [dict(r) for r in conn.execute("SELECT * FROM taxa WHERE parent_id=? AND taxid!=? ORDER BY total_count DESC, taxid", (parent, parent))]
if offset and offset >= len(rows):
raise ValueError("This group has no further lineages.")
shown, rest = rows[offset:offset + COLUMN_LIMIT], rows[offset + COLUMN_LIMIT:]
pinned = [r for r in rest if selected and not selected[1] and r['taxid'] == selected[0]]
shown, rest = shown + pinned, [r for r in rest if r not in pinned]
nodes = [node(r) for r in shown]
if rest:
nodes.append(remainder(rest, parent, offset + COLUMN_LIMIT))
row = taxon(parent)
if not offset and row['direct_count'] and rows:
nodes.append(dict(taxid=parent, offset=0, name="Assemblies assigned directly to this taxon",
label="Direct assignments", english="", icon="", borrowed_icon=False, total=row['direct_count'],
covered=row.get('covered_direct', 0), direct=True))
return nodes
columns = [[node(taxon(EUKARYOTA))]]
for depth, step in enumerate(steps[1:], 1):
if step[1] and (step[0] != steps[depth - 1][0] or step[1] <= steps[depth - 1][1]) or \
not step[1] and taxon(step[0])['parent_id'] != steps[depth - 1][0]:
raise ValueError("Choose a eukaryotic taxon from this snapshot.")
columns.append(column(steps[depth - 1], step))
tail = column(steps[-1], None)
if tail:
columns.append(tail)
for depth, nodes in enumerate(columns):
for item in nodes:
item['path'] = encode_path(steps[:depth] + [(item['taxid'], item['offset'])])
item['selected'] = depth < len(steps) and not item.get('direct') and \
(item['taxid'], item['offset']) == steps[depth]
if depth < len(steps) and not any(item['selected'] for item in nodes):
raise ValueError("Choose a eukaryotic taxon from this snapshot.")
lineage = [taxon(t) for t, _ in steps]
shortcuts = {t: encode_path((r['taxid'], 0) for r in self.lineage(conn, t)) for t, _ in SHORTCUTS}
return self.render(columns, steps, lineage, shortcuts, has_children=bool(tail))
def render(self, columns, steps, lineage, shortcuts, has_children):
"""Levels run down the page, so a whole lineage is one page scroll.
Each level is a horizontal band: bar widths follow assembly counts and
the caption sits above its bar, painted over the ribbons arriving there
rather than breaking them. Depth costs height, which the page can always
give, rather than width, which it cannot.
"""
left, span, bar, top = 16, 1180, 15, 10
rung = 152 # distance from one level's caption to the next
line = 15
# Silhouettes are not square: they run from tall and thin to long and
# flat. Fitting them all into one box turns most into slivers, so each
# gets its own box with the same area and its own proportions.
icon_area, icon_max, icon_band = 24 * 24, 42, 46
caption = icon_band + 2 * line # silhouette, scientific name, English name
width = left * 2 + span
height = top + (len(columns) - 1) * rung + caption + bar + 12
def spread(nodes):
"""Bar widths follow assembly counts, with room above each for its caption."""
gap, slot = (14 if len(nodes) > 1 else 0), 148
fixed, scale = set(), 0
for _ in range(len(nodes) + 1):
free = span - gap * (len(nodes) - 1) - slot * len(fixed)
flexible = sum(n['total'] for k, n in enumerate(nodes) if k not in fixed)
scale = free / flexible if flexible else 0
small = fixed | {k for k, n in enumerate(nodes) if n['total'] * scale < slot}
if small == fixed:
break
fixed = small
x = left
for k, n in enumerate(nodes):
room = slot if k in fixed else n['total'] * scale
n['w'] = max(3, n['total'] * scale)
n['x'] = x + (room - n['w']) / 2
n['slot'], n['slot_x'] = room, x
x += room + gap
def clip(text, limit):
return text if len(text) <= limit else text[:limit - 2] + "…"
def share(n):
return n['covered'] / n['total'] if self.coverage and n['total'] else 0
def describe(n):
"""The counts live here rather than in the chart, which stays uncluttered."""
percent = 100 * share(n)
title = f"{n['name']} ({n['english']})" if n['english'] else n['name']
detail = (f"{title} · {n['covered']:,} of {n['total']:,} assemblies have published annotations ({percent:.2f}%). Includes partial assemblies."
if self.coverage else f"{title} · {n['total']:,} assemblies. Coverage inventory unavailable.")
if n.get('aggregate'):
detail += " Display aggregate, not a taxonomic clade. Activate to see its lineages."
elif not n.get('direct') and not n['selected']:
detail += " Activate to open this group below."
return detail
def ribbon(y0, a0, a1, y1, b0, b1, kind, n):
bend = (y0 + y1) / 2
# The flux into a selected group is the trunk you have walked, so it
# stays emphasised rather than waiting for a hover.
trunk = " is-path" if n['selected'] else ""
return (f'<path d="M{a0:.1f},{y0} C{a0:.1f},{bend} {b0:.1f},{bend} {b0:.1f},{y1} '
f'L{b1:.1f},{y1} C{b1:.1f},{bend} {a1:.1f},{bend} {a1:.1f},{y0} Z" class="flow flow-{kind}{trunk}" '
f'data-node="{n["uid"]}" data-detail="{escape(n["detail"], quote=True)}"/>')
for depth, column in enumerate(columns):
spread(column)
for index, n in enumerate(column):
n['y'] = top + depth * rung # top of the caption
n['bar_y'] = n['y'] + caption # top of the bar
n['uid'] = f"{depth}-{index}"
n['detail'] = describe(n)
flows, nodes = [], []
for depth, column in enumerate(columns[:-1]):
parent = next((n for n in column if n['selected']), None)
children = columns[depth + 1]
if not parent:
continue
total = sum(c['total'] for c in children) or 1
cursor = parent['x']
for child in children:
a0 = cursor
a1 = cursor + parent['w'] * child['total'] / total
cursor = a1
f = share(child)
am, bm = a0 + f * (a1 - a0), child['x'] + f * child['w']
y0, y1 = parent['bar_y'] + bar, child['bar_y']
if f > 0:
flows.append(ribbon(y0, a0, am, y1, child['x'], bm, "annotated", child))
if f < 1:
flows.append(ribbon(y0, am, a1, y1, bm, child['x'] + child['w'], "missing", child))
for depth, column in enumerate(columns):
on_path = depth < len(steps)
shown = Counter(n['icon'] for n in column if n['icon'])
for n in column:
if n['borrowed_icon'] and shown[n['icon']] > 1:
n['icon'] = "" # identical borrowed outlines read as "same group"
x, y, w, bar_y = n['x'], n['y'], n['w'], n['bar_y']
f = share(n)
action = "" if n.get('direct') else f'data-path="{n["path"]}" role="button" tabindex="0"'
classes = "tree-node" + (" is-selected" if n['selected'] else " is-sibling" if on_path else "") + \
(" is-direct" if n.get('direct') else "")
nodes.append(f'<g class="{classes}" {action} data-node="{n["uid"]}" '
f'aria-label="{escape(n["detail"], quote=True)}" data-detail="{escape(n["detail"], quote=True)}">')
centre = n['slot_x'] + n['slot'] / 2
nodes.append(f'<rect x="{n["slot_x"] - 5:.1f}" y="{y - 4}" width="{n["slot"] + 10:.1f}" '
f'height="{caption + bar + 8:.0f}" rx="12" class="tree-hit"/>')
green = w * f
if green > 0:
nodes.append(f'<rect x="{x:.1f}" y="{bar_y}" width="{green:.1f}" height="{bar}" class="bar bar-annotated"/>')
if green < w:
nodes.append(f'<rect x="{x + green:.1f}" y="{bar_y}" width="{w - green:.1f}" height="{bar}" class="bar bar-missing"/>')
nodes.append(f'<rect x="{x:.1f}" y="{bar_y}" width="{w:.1f}" height="{bar}" class="bar-outline"/>')
label, english = clip(n['label'], MAX_LABEL), clip(n['english'], MAX_ENGLISH)
# The silhouette band is reserved whether or not this node has one,
# so names sit on the same baseline right across a level.
if n['icon']:
ratio = self.icon_credits.get(n['icon'], {}).get('ratio') or 1.0
icon_h = min(icon_max, max(6, math.sqrt(icon_area / ratio)))
icon_w = min(icon_max, max(6, icon_h * ratio))
icon_h = min(icon_max, max(6, icon_w / ratio))
nodes.append(f'<image class="tree-icon" x="{centre - icon_w / 2:.1f}" '
f'y="{y + (icon_band - icon_h) / 2:.1f}" '
f'width="{icon_w:.1f}" height="{icon_h:.1f}" '
f'href="{ICON_ROUTE}{ICON_DIR / (n["icon"] + ".svg")}"/>')
nodes.append(f'<text x="{centre:.1f}" y="{y + icon_band + line - 4:.0f}" class="tree-label">{escape(label)}</text>')
if english:
nodes.append(f'<text x="{centre:.1f}" y="{y + icon_band + 2 * line - 4:.0f}" class="tree-english">{escape(english)}</text>')
nodes.append('</g>')
svg = (f'<svg viewBox="0 0 {width} {height}" class="life-tree" data-route="{encode_path(steps)}" '
f'aria-label="Eukaryotic taxonomy and annotation coverage">' + ''.join(flows) + ''.join(nodes) + '</svg>')
focus = next(n for n in columns[len(steps) - 1] if n['selected'])
total, covered = focus['total'], focus['covered']
percent = 100 * covered / total if self.coverage and total else 0
selected_name = f"Other lineages of {lineage[-1]['name']}" if focus.get('aggregate') else focus['label']
if focus.get('english'):
selected_name += f" · {focus['english']}"
stamp = (self.coverage or self.metadata)['created_at'][:10]
if self.coverage:
summary = (f'<div class="atlas-kicker">{escape(selected_name)} · ASSEMBLY COVERAGE</div>'
f'<div class="atlas-summary"><strong>{covered:,}</strong><span>of {total:,}<br>assemblies annotated</span>'
f'<b class="atlas-badge">{percent:.1f}%</b></div>')
legend = ('<div class="atlas-legend"><i class="atlas-swatch atlas-annotated"></i><span>Annotated</span>'
'<i class="atlas-swatch atlas-missing"></i><span>Not annotated</span></div>')
else:
summary = f'<div class="atlas-kicker">{escape(selected_name)}</div><div class="atlas-summary"><strong>{total:,}</strong><span>GenBank assemblies<br>Coverage unavailable</span></div>'
legend = '<div class="atlas-legend"><i class="atlas-swatch atlas-missing"></i><span>Assemblies · coverage unavailable</span></div>'
jumps = ''.join(f'<button data-path="{shortcuts[t]}">{escape(name)}</button>' for t, name in SHORTCUTS)
leaf_note = '<p class="atlas-leaf-note">This is a terminal taxon in the assembly snapshot. Pick a level above to explore its relatives.</p>' if not has_children else ''
return f'''<section class="atlas" aria-label="Tree of eukaryotic life">
<header class="atlas-header">
<div><h1>The tree of eukaryotic life<span>.</span></h1>
<p>Follow the branches. Discover where we’ve annotated.</p></div>
<div class="atlas-summary-block">{summary}</div>
</header>
<nav class="atlas-nav" aria-label="Explore major lineages">
<div class="atlas-shortcuts">{jumps}</div>
</nav>
<div class="tree-scroll" tabindex="0" aria-label="Interactive tree. Expanding a group adds a level below it.">{svg}</div>
{leaf_note}<div class="atlas-tooltip" role="tooltip" hidden></div>
<footer class="atlas-footer">{legend}
<span>Click a group to open it below</span>
<span class="atlas-date">Updated {stamp}</span></footer>
<div class="atlas-footnote">Bar and flow widths are proportional to assembly counts within each level; each level is rescaled to fill the width, so compare sizes within a level. Coverage counts assemblies with published annotations, including partial assemblies.</div>
</section>'''
def credit_lines(taxonomy):
"""One line per artist, as CC BY asks, with the licences their images carry."""
artists = {}
for image in set(taxonomy.icons.values()):
meta = taxonomy.icon_credits.get(image, {})
name = meta.get("attribution") or meta.get("contributor") or "Unknown artist"
entry = artists.setdefault(name, {"licenses": set(), "pages": []})
entry["licenses"].add(LICENSE_NAMES.get(meta.get("license"), meta.get("license") or "unknown licence"))
entry["pages"].append(meta.get("page"))
lines = []
for name in sorted(artists, key=str.casefold):
entry = artists[name]
count = len(entry["pages"])
lines.append(f"- [{name}]({entry['pages'][0]}) — {', '.join(sorted(entry['licenses']))}"
f"{f' — {count} silhouettes' if count > 1 else ''}")
return lines
def build_taxonomy_tab():
"""Render the tree and its controls, and hand back the jump into the Database tab."""
if not DATABASE.exists():
gr.Markdown("The eukaryotic tree will appear when the taxonomy snapshot is available.")
return None
taxonomy = Taxonomy()
if taxonomy.icons:
# The chart references each silhouette by URL, so the browser caches it
# once instead of the file riding along in every re-render.
gr.set_static_paths(paths=[ICON_DIR])
tree = gr.HTML(taxonomy.view(), css_template=(ROOT / "atlas.css").read_text(),
js_on_load=(ROOT / "atlas.js").read_text(), apply_default_css=False,
elem_id="eukaryotic-atlas")
route = gr.State(DEFAULT_PATH)
def jump_text(path):
# The arrow marks it as leaving the atlas for the other view.
name = taxonomy.selected_name(path)
return (f"Show database results for ‘{name}’ ↗" if taxonomy.assemblies_for(path)
else f"No annotated assemblies in ‘{name}’")
def jump_label(path):
return gr.Button(jump_text(path), interactive=bool(taxonomy.assemblies_for(path)))
# The jump into the Database tab needs a real accession, so the button says
# up front whether the current group has any annotated assembly behind it.
open_database = gr.Button(jump_text(DEFAULT_PATH), variant="primary",
elem_classes="action-button", elem_id="atlas-open-database")
def show(path):
return taxonomy.view(path), path, jump_label(path)
def explore_tree(evt: gr.EventData):
try:
return show(evt.route)
except (ValueError, TypeError, AttributeError) as exc:
raise gr.Error("Choose a eukaryotic group in this snapshot.") from exc
tree.click(explore_tree, outputs=[tree, route, open_database], show_progress="minimal", concurrency_limit=1)
with gr.Accordion("Find a lineage", open=False, elem_classes="atlas-disclosure"):
query = gr.Textbox(label="Find a lineage", show_label=False, elem_classes="lineage-query",
placeholder="Scientific name, English name such as sponges or jellyfish, or a taxon ID")
status = gr.Markdown("Start typing to see matching lineages.", elem_classes="quiet-note")
matches = gr.Radio(choices=[], label="Matching lineages", show_label=False,
interactive=True, elem_classes="lineage-matches")
def taxonomy_search(text):
choices = taxonomy.search(text)
if not choices:
typed = str(text or "").strip()
return gr.Radio(choices=[], value=None), (
"No matching eukaryotic taxa in this snapshot." if len(typed) >= 2
else "Start typing to see matching lineages.")
best = choices[0][0].split(" · ")[0]
return (gr.Radio(choices=choices, value=None),
f"{len(choices)} matches, closest first. Pick one to open it, or press Enter for **{best}**.")
def taxonomy_view(taxid):
return show(taxonomy.taxon_path(taxid) if taxid else DEFAULT_PATH)
def taxonomy_best(text):
"""Enter moves the chart, so it ticks the row it moved to as well.
Showing a lineage while the list it came from sits unselected reads
as two answers to the same question.
"""
choices = taxonomy.search(text)
if not choices:
return (*taxonomy_view(None), gr.Radio(choices=[], value=None),
"No matching eukaryotic taxa in this snapshot.")
taxid = choices[0][1]
return (*taxonomy_view(taxid), gr.Radio(choices=choices, value=taxid),
f"{len(choices)} matches, closest first. Showing **{choices[0][0].split(' · ')[0]}**.")
# No separate search button: the list follows what is typed, Enter takes
# the closest match, and picking any row opens it.
query.change(taxonomy_search, query, [matches, status], show_progress="hidden")
query.submit(taxonomy_best, query, [tree, route, open_database, matches, status])
matches.input(taxonomy_view, matches, [tree, route, open_database])
with gr.Accordion("About the tree and coverage", open=False, elem_classes="atlas-disclosure"):
gr.Markdown("This tree follows **NCBI Taxonomy within Eukaryota**. Animals and fungi share the Opisthokonta branch; "
"green plants include green algae. ‘Other lineages’ combines the remaining siblings for display and can be expanded. "
"Each expanded group adds a level below it with up to six direct lineages plus any remainder, so every ancestor stays visible "
"and a whole lineage reads as one page scroll. The opening view also expands the animal/fungal branch. Bars and flows "
"are drawn like a Sankey diagram: widths follow **actual assembly counts** within each level, split into annotated "
"and not-annotated assemblies.\n\n"
"Each group shows its scientific name with a plain-English name underneath where one is available. These come from "
"NCBI Taxonomy's common names plus a curated set of short glosses for the large unranked clades NCBI leaves unnamed, "
"such as Opisthokonta (animals and fungi) or Ecdysozoa (moulting animals). The glosses describe the living members of a "
"clade in everyday words; they are not formal taxonomic synonyms, so the scientific name stays the primary label.\n\n"
"Coverage means the fraction of current GenBank assembly versions with published annotations, including partial assemblies. "
"It does not measure the fraction of bases annotated. Parent counts include descendants once each. "
"Exact accession versions are matched. The Database tab searches a dated snapshot of all published annotation files.\n\n"
"Sources: [NCBI assembly summary](https://ftp.ncbi.nlm.nih.gov/genomes/ASSEMBLY_REPORTS/assembly_summary_genbank.txt), "
"[NCBI Taxonomy](https://ftp.ncbi.nlm.nih.gov/pub/taxonomy/taxdump.tar.gz) for both the tree and its common names, "
"and [published annotations](https://huggingface.co/buckets/HuggingFaceBio/genbank-annotations).")
if taxonomy.icons:
gr.Markdown("Silhouettes come from [PhyloPic](https://www.phylopic.org), resolved from each taxon's NCBI ID and "
"stored unmodified. Only public-domain (CC0) and CC BY 4.0 images are used; NonCommercial and "
"ShareAlike images are filtered out at build time. A group without a silhouette of its own borrows "
"its nearest illustrated ancestor, so a fungal genus shows the fungal outline.")
coverage = taxonomy.coverage or {}
gr.JSON({"taxonomy_updated": taxonomy.metadata['created_at'], "coverage_updated": coverage.get('created_at'),
"silhouettes": taxonomy.icon_meta or None,
"display_scope": "Eukaryota (NCBI taxid 2759) and its descendants only",
"inventory_method": coverage.get('method'),
"published_versions_outside_current_GenBank_snapshot": coverage.get('unmatched_assembly_versions'),
"sources": taxonomy.metadata['sources']}, label="Snapshot provenance")
if taxonomy.icons:
with gr.Accordion("Silhouette credits", open=False, elem_classes="atlas-disclosure"):
gr.Markdown("\n".join(["Silhouettes from [PhyloPic](https://www.phylopic.org), used unmodified. "
"Public-domain images need no credit; the CC BY ones do, and every artist is listed here. "
"`data/icons.json` maps each taxon to its image.", ""] + credit_lines(taxonomy)))
def first_accession(path):
found = taxonomy.assemblies_for(path)
return found[0] if found else ""
return {"route": route, "button": open_database, "accession_for": first_accession}