Download taxonomy.py from HuggingFaceBio/carbon-a-database-explorer: direct link, hf CLI and curl.
- Browser
- Download file 33.5 kB
-
https://huggingface.co/spaces/HuggingFaceBio/carbon-a-database-explorer/resolve/main/taxonomy.py
- Command line
-
hf download hf://spaces/HuggingFaceBio/carbon-a-database-explorer/taxonomy.py
-
curl -L -o taxonomy.py https://huggingface.co/spaces/HuggingFaceBio/carbon-a-database-explorer/resolve/main/taxonomy.py
33.5 kB
| """Read-only taxonomy navigation, independent of the annotation lookup index.""" | |
| from collections import Counter | |
| from contextlib import closing | |
| import math | |
| from html import escape | |
| import json | |
| from pathlib import Path | |
| import sqlite3 | |
| import gradio as gr | |
| ROOT = Path(__file__).resolve().parent | |
| DATABASE = ROOT / "data/taxonomy.sqlite" | |
| COMMON_NAMES = ROOT / "data/common_names.json" | |
| ICONS = ROOT / "data/icons.json" | |
| ICON_DIR = ROOT / "data/icons" | |
| # Gradio serves whitelisted files from this route; see build_taxonomy_tab. | |
| ICON_ROUTE = "/gradio_api/file=" | |
| LICENSE_NAMES = {"https://creativecommons.org/publicdomain/zero/1.0/": "CC0 1.0", | |
| "https://creativecommons.org/licenses/by/4.0/": "CC BY 4.0", | |
| "https://creativecommons.org/licenses/by/3.0/": "CC BY 3.0", | |
| "https://creativecommons.org/publicdomain/mark/1.0/": "Public Domain Mark 1.0"} | |
| EUKARYOTA = 2759 | |
| EUK_TREE = """WITH RECURSIVE euk(taxid) AS ( | |
| SELECT 2759 UNION ALL | |
| SELECT t.taxid FROM taxa t JOIN euk e ON t.parent_id=e.taxid | |
| WHERE t.taxid != t.parent_id | |
| ) """ | |
| # Scientific names stay the primary label; English glosses are a secondary line. | |
| # See refresh_common_names.py for how the lookup is built. | |
| MAX_ENGLISH = 28 | |
| MAX_LABEL = 22 | |
| # Jumping-off points above the tree. Each one expands its whole lineage, so a | |
| # deep pick like humans is also the quickest way to see what a full path looks | |
| # like. Order runs broad to narrow. | |
| SHORTCUTS = ((33208, "Animals"), (40674, "Mammals"), (9606, "Humans"), (8782, "Birds"), | |
| (7898, "Ray-finned fish"), (50557, "Insects"), (6142, "Jellyfish"), | |
| (33090, "Green plants"), (4751, "Fungi")) | |
| COLUMN_LIMIT = 6 | |
| SEARCH_LIMIT = 12 | |
| MAX_DEPTH = 80 | |
| DEFAULT_PATH = "2759/33154" | |
| def encode_path(steps): | |
| """Serialize selected (taxid, offset) pairs; an offset marks an expanded "Other lineages" aggregate.""" | |
| return "/".join(f"{t}~{o}" if o else str(t) for t, o in steps) | |
| class Taxonomy: | |
| def __init__(self, path=DATABASE, common_names=COMMON_NAMES, icons=ICONS): | |
| self.path = Path(path) | |
| self.english, self.english_sources = {}, {} | |
| if Path(common_names).exists(): | |
| payload = json.loads(Path(common_names).read_text()) | |
| self.english = {int(t): name for t, name in payload["names"].items()} | |
| self.english_sources = payload.get("sources", {}) | |
| self.icons, self.icon_credits, self.icon_meta = {}, {}, {} | |
| if Path(icons).exists(): | |
| payload = json.loads(Path(icons).read_text()) | |
| self.icon_meta = {k: v for k, v in payload.items() if k not in ("images", "taxa")} | |
| self.icon_credits = payload.get("images", {}) | |
| self.icons = {int(t): image for t, image in payload.get("taxa", {}).items() | |
| if (ICON_DIR / f"{image}.svg").exists()} | |
| self._inherited_icons = {} | |
| with closing(self.connect()) as conn: | |
| self.metadata = json.loads(conn.execute("SELECT value FROM metadata").fetchone()[0]) | |
| self.eukaryote_ids = frozenset(r[0] for r in conn.execute(EUK_TREE + "SELECT taxid FROM euk")) | |
| self.eukaryotes = dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (EUKARYOTA,)).fetchone()) | |
| self.coverage = self.metadata.get("coverage") | |
| def connect(self): | |
| conn = sqlite3.connect(f"{self.path.resolve().as_uri()}?mode=ro", uri=True) | |
| conn.row_factory = sqlite3.Row | |
| return conn | |
| def closeness(self, row, needle): | |
| """Whole-name matches first, then word starts: 'thale' should find thale cress | |
| before it finds Magnaporthales.""" | |
| names = [row['name'].casefold(), self.english.get(row['taxid'], "").casefold()] | |
| if any(name == needle for name in names): | |
| return 0 | |
| if any(name.startswith(needle) for name in names): | |
| return 1 | |
| if any(word.startswith(needle) for name in names for word in name.split()): | |
| return 2 | |
| return 3 | |
| def label(self, row): | |
| english = self.english.get(row['taxid'], "") | |
| return (f"{row['name']}" + (f" · {english}" if english else "") + | |
| f" · {row['total_count']:,} assemblies · taxid {row['taxid']}") | |
| def search(self, query, limit=SEARCH_LIMIT): | |
| """Match a taxon id, a scientific name, or one of the English names. | |
| The eukaryote set is already in memory, so the filtering happens here | |
| rather than in a recursive CTE the query would otherwise re-walk on every | |
| keystroke. | |
| """ | |
| query = str(query or "").strip() | |
| if len(query) < 2 and not query.isdigit(): | |
| return [] | |
| needle = query.casefold() | |
| with closing(self.connect()) as conn: | |
| if query.lstrip("-").isdigit(): | |
| row = conn.execute("SELECT * FROM taxa WHERE taxid=?", (int(query),)).fetchone() | |
| rows = [row] if row and row['taxid'] in self.eukaryote_ids else [] | |
| else: | |
| pattern = "%" + query.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_") + "%" | |
| found = {r['taxid']: r for r in conn.execute( | |
| "SELECT * FROM taxa WHERE name LIKE ? ESCAPE '\\' ORDER BY total_count DESC LIMIT ?", | |
| (pattern, limit * 4)) if r['taxid'] in self.eukaryote_ids} | |
| english = [t for t, name in self.english.items() | |
| if needle in name.casefold() and t not in found and t in self.eukaryote_ids] | |
| for chunk in (english[i:i + 400] for i in range(0, len(english), 400)): | |
| marks = ",".join("?" * len(chunk)) | |
| found.update({r['taxid']: r for r in conn.execute( | |
| f"SELECT * FROM taxa WHERE taxid IN ({marks})", chunk)}) | |
| rows = sorted(found.values(), key=lambda r: (self.closeness(r, needle), -r['total_count']))[:limit] | |
| return [(self.label(r), str(r['taxid'])) for r in rows] | |
| def icon_for(self, conn, taxid): | |
| """Nearest icon at or above this taxon, following the lineage to Eukaryota. | |
| Returns the image and whether it belongs to this taxon or was borrowed. | |
| """ | |
| if taxid in self.icons: | |
| return self.icons[taxid], False | |
| if taxid in self._inherited_icons: | |
| return self._inherited_icons[taxid], True | |
| walked, current = [], taxid | |
| for _ in range(MAX_DEPTH): | |
| row = conn.execute("SELECT parent_id FROM taxa WHERE taxid=?", (current,)).fetchone() | |
| if row is None or row['parent_id'] == current: | |
| break | |
| walked.append(current) | |
| current = row['parent_id'] | |
| if current in self.icons or current in self._inherited_icons: | |
| break | |
| found = self.icons.get(current) or self._inherited_icons.get(current, "") | |
| for step in walked: | |
| self._inherited_icons[step] = found | |
| return found, True | |
| def lineage(self, conn, taxid): | |
| rows = [dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (taxid,)).fetchone())] | |
| while rows[-1]['taxid'] != EUKARYOTA: | |
| rows.append(dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (rows[-1]['parent_id'],)).fetchone())) | |
| return rows[::-1] | |
| def assemblies_for(self, path, limit=1): | |
| """Annotated accessions under the taxon a route selects, largest group first. | |
| The recursive walk starts at the selected taxon rather than at Eukaryota, | |
| so it touches only that subtree. | |
| """ | |
| try: | |
| taxid = int(str(path).split("/")[-1].partition("~")[0]) | |
| except (ValueError, AttributeError): | |
| return [] | |
| if taxid not in self.eukaryote_ids: | |
| return [] | |
| with closing(self.connect()) as conn: | |
| if not conn.execute("SELECT name FROM sqlite_master WHERE type='table' AND name='assemblies'").fetchone(): | |
| return [] | |
| return [r[0] for r in conn.execute( | |
| """WITH RECURSIVE sub(taxid) AS ( | |
| SELECT ? UNION ALL | |
| SELECT t.taxid FROM taxa t JOIN sub ON t.parent_id=sub.taxid WHERE t.taxid!=t.parent_id | |
| ) SELECT accession FROM assemblies JOIN sub USING(taxid) ORDER BY accession LIMIT ?""", | |
| (taxid, limit))] | |
| def selected_name(self, path): | |
| taxid = int(str(path).split("/")[-1].partition("~")[0]) | |
| with closing(self.connect()) as conn: | |
| row = conn.execute("SELECT name FROM taxa WHERE taxid=?", (taxid,)).fetchone() | |
| return row['name'] if row else str(taxid) | |
| def taxon_path(self, taxid): | |
| """Path that expands every ancestor of a taxon, for search results and shortcuts.""" | |
| taxid = int(taxid) | |
| if taxid not in self.eukaryote_ids: | |
| raise ValueError("Choose a eukaryotic taxon from this snapshot.") | |
| with closing(self.connect()) as conn: | |
| return encode_path((r['taxid'], 0) for r in self.lineage(conn, taxid)) | |
| def view_taxon(self, taxid): | |
| return self.view(self.taxon_path(taxid)) | |
| def view(self, path=DEFAULT_PATH): | |
| """Render one column per expanded group. A path lists the selected node of each column.""" | |
| try: | |
| steps = [(int(t), int(o or 0)) for t, _, o in (part.partition("~") for part in str(path).split("/"))] | |
| except ValueError as exc: | |
| raise ValueError("Choose a eukaryotic taxon from this snapshot.") from exc | |
| if steps[0] != (EUKARYOTA, 0) or len(steps) > MAX_DEPTH or any( | |
| t not in self.eukaryote_ids or o < 0 for t, o in steps): | |
| raise ValueError("Choose a eukaryotic taxon from this snapshot.") | |
| with closing(self.connect()) as conn: | |
| def taxon(t): | |
| return dict(conn.execute("SELECT * FROM taxa WHERE taxid=?", (t,)).fetchone()) | |
| def node(row, offset=0, **extra): | |
| return dict(taxid=row['taxid'], offset=offset, name=row['name'], label=row['name'], | |
| english=self.english.get(row['taxid'], ""), | |
| **dict(zip(("icon", "borrowed_icon"), self.icon_for(conn, row['taxid']))), | |
| total=row['total_count'], covered=row.get('covered_total', 0), **extra) | |
| def remainder(rows, parent, start): | |
| return dict(taxid=parent, offset=start, name=f"Other lineages · {len(rows):,} groups (display aggregate)", | |
| label=f"Other lineages ({len(rows):,})", english="", icon="", borrowed_icon=False, total=sum(r['total_count'] for r in rows), | |
| covered=sum(r.get('covered_total', 0) for r in rows), aggregate=True) | |
| def column(step, selected): | |
| """Direct lineages of a selected node, keeping the next selection visible.""" | |
| parent, offset = step | |
| rows = [dict(r) for r in conn.execute("SELECT * FROM taxa WHERE parent_id=? AND taxid!=? ORDER BY total_count DESC, taxid", (parent, parent))] | |
| if offset and offset >= len(rows): | |
| raise ValueError("This group has no further lineages.") | |
| shown, rest = rows[offset:offset + COLUMN_LIMIT], rows[offset + COLUMN_LIMIT:] | |
| pinned = [r for r in rest if selected and not selected[1] and r['taxid'] == selected[0]] | |
| shown, rest = shown + pinned, [r for r in rest if r not in pinned] | |
| nodes = [node(r) for r in shown] | |
| if rest: | |
| nodes.append(remainder(rest, parent, offset + COLUMN_LIMIT)) | |
| row = taxon(parent) | |
| if not offset and row['direct_count'] and rows: | |
| nodes.append(dict(taxid=parent, offset=0, name="Assemblies assigned directly to this taxon", | |
| label="Direct assignments", english="", icon="", borrowed_icon=False, total=row['direct_count'], | |
| covered=row.get('covered_direct', 0), direct=True)) | |
| return nodes | |
| columns = [[node(taxon(EUKARYOTA))]] | |
| for depth, step in enumerate(steps[1:], 1): | |
| if step[1] and (step[0] != steps[depth - 1][0] or step[1] <= steps[depth - 1][1]) or \ | |
| not step[1] and taxon(step[0])['parent_id'] != steps[depth - 1][0]: | |
| raise ValueError("Choose a eukaryotic taxon from this snapshot.") | |
| columns.append(column(steps[depth - 1], step)) | |
| tail = column(steps[-1], None) | |
| if tail: | |
| columns.append(tail) | |
| for depth, nodes in enumerate(columns): | |
| for item in nodes: | |
| item['path'] = encode_path(steps[:depth] + [(item['taxid'], item['offset'])]) | |
| item['selected'] = depth < len(steps) and not item.get('direct') and \ | |
| (item['taxid'], item['offset']) == steps[depth] | |
| if depth < len(steps) and not any(item['selected'] for item in nodes): | |
| raise ValueError("Choose a eukaryotic taxon from this snapshot.") | |
| lineage = [taxon(t) for t, _ in steps] | |
| shortcuts = {t: encode_path((r['taxid'], 0) for r in self.lineage(conn, t)) for t, _ in SHORTCUTS} | |
| return self.render(columns, steps, lineage, shortcuts, has_children=bool(tail)) | |
| def render(self, columns, steps, lineage, shortcuts, has_children): | |
| """Levels run down the page, so a whole lineage is one page scroll. | |
| Each level is a horizontal band: bar widths follow assembly counts and | |
| the caption sits above its bar, painted over the ribbons arriving there | |
| rather than breaking them. Depth costs height, which the page can always | |
| give, rather than width, which it cannot. | |
| """ | |
| left, span, bar, top = 16, 1180, 15, 10 | |
| rung = 152 # distance from one level's caption to the next | |
| line = 15 | |
| # Silhouettes are not square: they run from tall and thin to long and | |
| # flat. Fitting them all into one box turns most into slivers, so each | |
| # gets its own box with the same area and its own proportions. | |
| icon_area, icon_max, icon_band = 24 * 24, 42, 46 | |
| caption = icon_band + 2 * line # silhouette, scientific name, English name | |
| width = left * 2 + span | |
| height = top + (len(columns) - 1) * rung + caption + bar + 12 | |
| def spread(nodes): | |
| """Bar widths follow assembly counts, with room above each for its caption.""" | |
| gap, slot = (14 if len(nodes) > 1 else 0), 148 | |
| fixed, scale = set(), 0 | |
| for _ in range(len(nodes) + 1): | |
| free = span - gap * (len(nodes) - 1) - slot * len(fixed) | |
| flexible = sum(n['total'] for k, n in enumerate(nodes) if k not in fixed) | |
| scale = free / flexible if flexible else 0 | |
| small = fixed | {k for k, n in enumerate(nodes) if n['total'] * scale < slot} | |
| if small == fixed: | |
| break | |
| fixed = small | |
| x = left | |
| for k, n in enumerate(nodes): | |
| room = slot if k in fixed else n['total'] * scale | |
| n['w'] = max(3, n['total'] * scale) | |
| n['x'] = x + (room - n['w']) / 2 | |
| n['slot'], n['slot_x'] = room, x | |
| x += room + gap | |
| def clip(text, limit): | |
| return text if len(text) <= limit else text[:limit - 2] + "…" | |
| def share(n): | |
| return n['covered'] / n['total'] if self.coverage and n['total'] else 0 | |
| def describe(n): | |
| """The counts live here rather than in the chart, which stays uncluttered.""" | |
| percent = 100 * share(n) | |
| title = f"{n['name']} ({n['english']})" if n['english'] else n['name'] | |
| detail = (f"{title} · {n['covered']:,} of {n['total']:,} assemblies have published annotations ({percent:.2f}%). Includes partial assemblies." | |
| if self.coverage else f"{title} · {n['total']:,} assemblies. Coverage inventory unavailable.") | |
| if n.get('aggregate'): | |
| detail += " Display aggregate, not a taxonomic clade. Activate to see its lineages." | |
| elif not n.get('direct') and not n['selected']: | |
| detail += " Activate to open this group below." | |
| return detail | |
| def ribbon(y0, a0, a1, y1, b0, b1, kind, n): | |
| bend = (y0 + y1) / 2 | |
| # The flux into a selected group is the trunk you have walked, so it | |
| # stays emphasised rather than waiting for a hover. | |
| trunk = " is-path" if n['selected'] else "" | |
| return (f'<path d="M{a0:.1f},{y0} C{a0:.1f},{bend} {b0:.1f},{bend} {b0:.1f},{y1} ' | |
| f'L{b1:.1f},{y1} C{b1:.1f},{bend} {a1:.1f},{bend} {a1:.1f},{y0} Z" class="flow flow-{kind}{trunk}" ' | |
| f'data-node="{n["uid"]}" data-detail="{escape(n["detail"], quote=True)}"/>') | |
| for depth, column in enumerate(columns): | |
| spread(column) | |
| for index, n in enumerate(column): | |
| n['y'] = top + depth * rung # top of the caption | |
| n['bar_y'] = n['y'] + caption # top of the bar | |
| n['uid'] = f"{depth}-{index}" | |
| n['detail'] = describe(n) | |
| flows, nodes = [], [] | |
| for depth, column in enumerate(columns[:-1]): | |
| parent = next((n for n in column if n['selected']), None) | |
| children = columns[depth + 1] | |
| if not parent: | |
| continue | |
| total = sum(c['total'] for c in children) or 1 | |
| cursor = parent['x'] | |
| for child in children: | |
| a0 = cursor | |
| a1 = cursor + parent['w'] * child['total'] / total | |
| cursor = a1 | |
| f = share(child) | |
| am, bm = a0 + f * (a1 - a0), child['x'] + f * child['w'] | |
| y0, y1 = parent['bar_y'] + bar, child['bar_y'] | |
| if f > 0: | |
| flows.append(ribbon(y0, a0, am, y1, child['x'], bm, "annotated", child)) | |
| if f < 1: | |
| flows.append(ribbon(y0, am, a1, y1, bm, child['x'] + child['w'], "missing", child)) | |
| for depth, column in enumerate(columns): | |
| on_path = depth < len(steps) | |
| shown = Counter(n['icon'] for n in column if n['icon']) | |
| for n in column: | |
| if n['borrowed_icon'] and shown[n['icon']] > 1: | |
| n['icon'] = "" # identical borrowed outlines read as "same group" | |
| x, y, w, bar_y = n['x'], n['y'], n['w'], n['bar_y'] | |
| f = share(n) | |
| action = "" if n.get('direct') else f'data-path="{n["path"]}" role="button" tabindex="0"' | |
| classes = "tree-node" + (" is-selected" if n['selected'] else " is-sibling" if on_path else "") + \ | |
| (" is-direct" if n.get('direct') else "") | |
| nodes.append(f'<g class="{classes}" {action} data-node="{n["uid"]}" ' | |
| f'aria-label="{escape(n["detail"], quote=True)}" data-detail="{escape(n["detail"], quote=True)}">') | |
| centre = n['slot_x'] + n['slot'] / 2 | |
| nodes.append(f'<rect x="{n["slot_x"] - 5:.1f}" y="{y - 4}" width="{n["slot"] + 10:.1f}" ' | |
| f'height="{caption + bar + 8:.0f}" rx="12" class="tree-hit"/>') | |
| green = w * f | |
| if green > 0: | |
| nodes.append(f'<rect x="{x:.1f}" y="{bar_y}" width="{green:.1f}" height="{bar}" class="bar bar-annotated"/>') | |
| if green < w: | |
| nodes.append(f'<rect x="{x + green:.1f}" y="{bar_y}" width="{w - green:.1f}" height="{bar}" class="bar bar-missing"/>') | |
| nodes.append(f'<rect x="{x:.1f}" y="{bar_y}" width="{w:.1f}" height="{bar}" class="bar-outline"/>') | |
| label, english = clip(n['label'], MAX_LABEL), clip(n['english'], MAX_ENGLISH) | |
| # The silhouette band is reserved whether or not this node has one, | |
| # so names sit on the same baseline right across a level. | |
| if n['icon']: | |
| ratio = self.icon_credits.get(n['icon'], {}).get('ratio') or 1.0 | |
| icon_h = min(icon_max, max(6, math.sqrt(icon_area / ratio))) | |
| icon_w = min(icon_max, max(6, icon_h * ratio)) | |
| icon_h = min(icon_max, max(6, icon_w / ratio)) | |
| nodes.append(f'<image class="tree-icon" x="{centre - icon_w / 2:.1f}" ' | |
| f'y="{y + (icon_band - icon_h) / 2:.1f}" ' | |
| f'width="{icon_w:.1f}" height="{icon_h:.1f}" ' | |
| f'href="{ICON_ROUTE}{ICON_DIR / (n["icon"] + ".svg")}"/>') | |
| nodes.append(f'<text x="{centre:.1f}" y="{y + icon_band + line - 4:.0f}" class="tree-label">{escape(label)}</text>') | |
| if english: | |
| nodes.append(f'<text x="{centre:.1f}" y="{y + icon_band + 2 * line - 4:.0f}" class="tree-english">{escape(english)}</text>') | |
| nodes.append('</g>') | |
| svg = (f'<svg viewBox="0 0 {width} {height}" class="life-tree" data-route="{encode_path(steps)}" ' | |
| f'aria-label="Eukaryotic taxonomy and annotation coverage">' + ''.join(flows) + ''.join(nodes) + '</svg>') | |
| focus = next(n for n in columns[len(steps) - 1] if n['selected']) | |
| total, covered = focus['total'], focus['covered'] | |
| percent = 100 * covered / total if self.coverage and total else 0 | |
| selected_name = f"Other lineages of {lineage[-1]['name']}" if focus.get('aggregate') else focus['label'] | |
| if focus.get('english'): | |
| selected_name += f" · {focus['english']}" | |
| stamp = (self.coverage or self.metadata)['created_at'][:10] | |
| if self.coverage: | |
| summary = (f'<div class="atlas-kicker">{escape(selected_name)} · ASSEMBLY COVERAGE</div>' | |
| f'<div class="atlas-summary"><strong>{covered:,}</strong><span>of {total:,}<br>assemblies annotated</span>' | |
| f'<b class="atlas-badge">{percent:.1f}%</b></div>') | |
| legend = ('<div class="atlas-legend"><i class="atlas-swatch atlas-annotated"></i><span>Annotated</span>' | |
| '<i class="atlas-swatch atlas-missing"></i><span>Not annotated</span></div>') | |
| else: | |
| summary = f'<div class="atlas-kicker">{escape(selected_name)}</div><div class="atlas-summary"><strong>{total:,}</strong><span>GenBank assemblies<br>Coverage unavailable</span></div>' | |
| legend = '<div class="atlas-legend"><i class="atlas-swatch atlas-missing"></i><span>Assemblies · coverage unavailable</span></div>' | |
| jumps = ''.join(f'<button data-path="{shortcuts[t]}">{escape(name)}</button>' for t, name in SHORTCUTS) | |
| leaf_note = '<p class="atlas-leaf-note">This is a terminal taxon in the assembly snapshot. Pick a level above to explore its relatives.</p>' if not has_children else '' | |
| return f'''<section class="atlas" aria-label="Tree of eukaryotic life"> | |
| <header class="atlas-header"> | |
| <div><h1>The tree of eukaryotic life<span>.</span></h1> | |
| <p>Follow the branches. Discover where we’ve annotated.</p></div> | |
| <div class="atlas-summary-block">{summary}</div> | |
| </header> | |
| <nav class="atlas-nav" aria-label="Explore major lineages"> | |
| <div class="atlas-shortcuts">{jumps}</div> | |
| </nav> | |
| <div class="tree-scroll" tabindex="0" aria-label="Interactive tree. Expanding a group adds a level below it.">{svg}</div> | |
| {leaf_note}<div class="atlas-tooltip" role="tooltip" hidden></div> | |
| <footer class="atlas-footer">{legend} | |
| <span>Click a group to open it below</span> | |
| <span class="atlas-date">Updated {stamp}</span></footer> | |
| <div class="atlas-footnote">Bar and flow widths are proportional to assembly counts within each level; each level is rescaled to fill the width, so compare sizes within a level. Coverage counts assemblies with published annotations, including partial assemblies.</div> | |
| </section>''' | |
| def credit_lines(taxonomy): | |
| """One line per artist, as CC BY asks, with the licences their images carry.""" | |
| artists = {} | |
| for image in set(taxonomy.icons.values()): | |
| meta = taxonomy.icon_credits.get(image, {}) | |
| name = meta.get("attribution") or meta.get("contributor") or "Unknown artist" | |
| entry = artists.setdefault(name, {"licenses": set(), "pages": []}) | |
| entry["licenses"].add(LICENSE_NAMES.get(meta.get("license"), meta.get("license") or "unknown licence")) | |
| entry["pages"].append(meta.get("page")) | |
| lines = [] | |
| for name in sorted(artists, key=str.casefold): | |
| entry = artists[name] | |
| count = len(entry["pages"]) | |
| lines.append(f"- [{name}]({entry['pages'][0]}) — {', '.join(sorted(entry['licenses']))}" | |
| f"{f' — {count} silhouettes' if count > 1 else ''}") | |
| return lines | |
| def build_taxonomy_tab(): | |
| """Render the tree and its controls, and hand back the jump into the Database tab.""" | |
| if not DATABASE.exists(): | |
| gr.Markdown("The eukaryotic tree will appear when the taxonomy snapshot is available.") | |
| return None | |
| taxonomy = Taxonomy() | |
| if taxonomy.icons: | |
| # The chart references each silhouette by URL, so the browser caches it | |
| # once instead of the file riding along in every re-render. | |
| gr.set_static_paths(paths=[ICON_DIR]) | |
| tree = gr.HTML(taxonomy.view(), css_template=(ROOT / "atlas.css").read_text(), | |
| js_on_load=(ROOT / "atlas.js").read_text(), apply_default_css=False, | |
| elem_id="eukaryotic-atlas") | |
| route = gr.State(DEFAULT_PATH) | |
| def jump_text(path): | |
| # The arrow marks it as leaving the atlas for the other view. | |
| name = taxonomy.selected_name(path) | |
| return (f"Show database results for ‘{name}’ ↗" if taxonomy.assemblies_for(path) | |
| else f"No annotated assemblies in ‘{name}’") | |
| def jump_label(path): | |
| return gr.Button(jump_text(path), interactive=bool(taxonomy.assemblies_for(path))) | |
| # The jump into the Database tab needs a real accession, so the button says | |
| # up front whether the current group has any annotated assembly behind it. | |
| open_database = gr.Button(jump_text(DEFAULT_PATH), variant="primary", | |
| elem_classes="action-button", elem_id="atlas-open-database") | |
| def show(path): | |
| return taxonomy.view(path), path, jump_label(path) | |
| def explore_tree(evt: gr.EventData): | |
| try: | |
| return show(evt.route) | |
| except (ValueError, TypeError, AttributeError) as exc: | |
| raise gr.Error("Choose a eukaryotic group in this snapshot.") from exc | |
| tree.click(explore_tree, outputs=[tree, route, open_database], show_progress="minimal", concurrency_limit=1) | |
| with gr.Accordion("Find a lineage", open=False, elem_classes="atlas-disclosure"): | |
| query = gr.Textbox(label="Find a lineage", show_label=False, elem_classes="lineage-query", | |
| placeholder="Scientific name, English name such as sponges or jellyfish, or a taxon ID") | |
| status = gr.Markdown("Start typing to see matching lineages.", elem_classes="quiet-note") | |
| matches = gr.Radio(choices=[], label="Matching lineages", show_label=False, | |
| interactive=True, elem_classes="lineage-matches") | |
| def taxonomy_search(text): | |
| choices = taxonomy.search(text) | |
| if not choices: | |
| typed = str(text or "").strip() | |
| return gr.Radio(choices=[], value=None), ( | |
| "No matching eukaryotic taxa in this snapshot." if len(typed) >= 2 | |
| else "Start typing to see matching lineages.") | |
| best = choices[0][0].split(" · ")[0] | |
| return (gr.Radio(choices=choices, value=None), | |
| f"{len(choices)} matches, closest first. Pick one to open it, or press Enter for **{best}**.") | |
| def taxonomy_view(taxid): | |
| return show(taxonomy.taxon_path(taxid) if taxid else DEFAULT_PATH) | |
| def taxonomy_best(text): | |
| """Enter moves the chart, so it ticks the row it moved to as well. | |
| Showing a lineage while the list it came from sits unselected reads | |
| as two answers to the same question. | |
| """ | |
| choices = taxonomy.search(text) | |
| if not choices: | |
| return (*taxonomy_view(None), gr.Radio(choices=[], value=None), | |
| "No matching eukaryotic taxa in this snapshot.") | |
| taxid = choices[0][1] | |
| return (*taxonomy_view(taxid), gr.Radio(choices=choices, value=taxid), | |
| f"{len(choices)} matches, closest first. Showing **{choices[0][0].split(' · ')[0]}**.") | |
| # No separate search button: the list follows what is typed, Enter takes | |
| # the closest match, and picking any row opens it. | |
| query.change(taxonomy_search, query, [matches, status], show_progress="hidden") | |
| query.submit(taxonomy_best, query, [tree, route, open_database, matches, status]) | |
| matches.input(taxonomy_view, matches, [tree, route, open_database]) | |
| with gr.Accordion("About the tree and coverage", open=False, elem_classes="atlas-disclosure"): | |
| gr.Markdown("This tree follows **NCBI Taxonomy within Eukaryota**. Animals and fungi share the Opisthokonta branch; " | |
| "green plants include green algae. ‘Other lineages’ combines the remaining siblings for display and can be expanded. " | |
| "Each expanded group adds a level below it with up to six direct lineages plus any remainder, so every ancestor stays visible " | |
| "and a whole lineage reads as one page scroll. The opening view also expands the animal/fungal branch. Bars and flows " | |
| "are drawn like a Sankey diagram: widths follow **actual assembly counts** within each level, split into annotated " | |
| "and not-annotated assemblies.\n\n" | |
| "Each group shows its scientific name with a plain-English name underneath where one is available. These come from " | |
| "NCBI Taxonomy's common names plus a curated set of short glosses for the large unranked clades NCBI leaves unnamed, " | |
| "such as Opisthokonta (animals and fungi) or Ecdysozoa (moulting animals). The glosses describe the living members of a " | |
| "clade in everyday words; they are not formal taxonomic synonyms, so the scientific name stays the primary label.\n\n" | |
| "Coverage means the fraction of current GenBank assembly versions with published annotations, including partial assemblies. " | |
| "It does not measure the fraction of bases annotated. Parent counts include descendants once each. " | |
| "Exact accession versions are matched. The Database tab searches a dated snapshot of all published annotation files.\n\n" | |
| "Sources: [NCBI assembly summary](https://ftp.ncbi.nlm.nih.gov/genomes/ASSEMBLY_REPORTS/assembly_summary_genbank.txt), " | |
| "[NCBI Taxonomy](https://ftp.ncbi.nlm.nih.gov/pub/taxonomy/taxdump.tar.gz) for both the tree and its common names, " | |
| "and [published annotations](https://huggingface.co/buckets/HuggingFaceBio/genbank-annotations).") | |
| if taxonomy.icons: | |
| gr.Markdown("Silhouettes come from [PhyloPic](https://www.phylopic.org), resolved from each taxon's NCBI ID and " | |
| "stored unmodified. Only public-domain (CC0) and CC BY 4.0 images are used; NonCommercial and " | |
| "ShareAlike images are filtered out at build time. A group without a silhouette of its own borrows " | |
| "its nearest illustrated ancestor, so a fungal genus shows the fungal outline.") | |
| coverage = taxonomy.coverage or {} | |
| gr.JSON({"taxonomy_updated": taxonomy.metadata['created_at'], "coverage_updated": coverage.get('created_at'), | |
| "silhouettes": taxonomy.icon_meta or None, | |
| "display_scope": "Eukaryota (NCBI taxid 2759) and its descendants only", | |
| "inventory_method": coverage.get('method'), | |
| "published_versions_outside_current_GenBank_snapshot": coverage.get('unmatched_assembly_versions'), | |
| "sources": taxonomy.metadata['sources']}, label="Snapshot provenance") | |
| if taxonomy.icons: | |
| with gr.Accordion("Silhouette credits", open=False, elem_classes="atlas-disclosure"): | |
| gr.Markdown("\n".join(["Silhouettes from [PhyloPic](https://www.phylopic.org), used unmodified. " | |
| "Public-domain images need no credit; the CC BY ones do, and every artist is listed here. " | |
| "`data/icons.json` maps each taxon to its image.", ""] + credit_lines(taxonomy))) | |
| def first_accession(path): | |
| found = taxonomy.assemblies_for(path) | |
| return found[0] if found else "" | |
| return {"route": route, "button": open_database, "accession_for": first_accession} | |