Download refresh_icons.py from HuggingFaceBio/carbon-a-database-explorer: direct link, hf CLI and curl.
- Browser
- Download file 11.6 kB
-
https://huggingface.co/spaces/HuggingFaceBio/carbon-a-database-explorer/resolve/main/refresh_icons.py
- Command line
-
hf download hf://spaces/HuggingFaceBio/carbon-a-database-explorer/refresh_icons.py
-
curl -L -o refresh_icons.py https://huggingface.co/spaces/HuggingFaceBio/carbon-a-database-explorer/resolve/main/refresh_icons.py
11.6 kB
| """Fetch clade silhouettes from PhyloPic for the groups the Genome Atlas shows. | |
| PhyloPic resolves NCBI taxids directly and walks up its own tree when a clade has | |
| no silhouette of its own, so a request for Eumetazoa answers with the nearest | |
| illustrated ancestor. Two filters are always applied: | |
| * ``filter_license_nc=false`` drops NonCommercial images. Whether a Space that | |
| sells nothing counts as noncommercial use is genuinely unsettled, and the | |
| question is avoidable: every group in the atlas has a freer alternative. | |
| * ``filter_license_sa=false`` drops ShareAlike images, whose terms would reach | |
| into the repository as soon as an icon were adapted. | |
| What remains is CC0 and CC BY 4.0. Files are stored exactly as PhyloPic serves | |
| them, unmodified, and ``data/icons.json`` carries the licence and attribution | |
| for each one so the interface can credit the artists. | |
| """ | |
| import argparse | |
| from collections import Counter | |
| import json | |
| import math | |
| from pathlib import Path | |
| import sqlite3 | |
| import time | |
| import urllib.error | |
| import urllib.request | |
| ROOT = Path(__file__).resolve().parent | |
| API = "https://api.phylopic.org" | |
| HEADERS = {"Accept": "application/vnd.phylopic.v2+json", | |
| "User-Agent": "huggingface-genome-atlas (https://huggingface.co/spaces/HuggingFaceBio/genbank-annotation-explorer)"} | |
| EUKARYOTA = 2759 | |
| COLUMN_LIMIT = 6 | |
| # Groups below this many assemblies inherit an ancestor's icon at render time. | |
| MIN_ASSEMBLIES = 100 | |
| # PhyloPic's representative image is usually right, but for a few flagship | |
| # groups it is accurate and still a poor mnemonic. Its image for the whole Fungi | |
| # kingdom is a chytrid zoospore: chytrids are early-diverging fungi and the only | |
| # ones that keep a flagellum, and that single posterior flagellum is the trait | |
| # Opisthokonta is named for, which is why it reads as an animal sperm cell. True | |
| # of the clade, useless as an icon, so a mushroom cluster stands in. | |
| OVERRIDES = {4751: "675ddef0-9dd4-4eef-b63f-98beb30f98c7"} | |
| # Columns within this many steps of the root get an icon regardless of size. | |
| SHALLOW_DEPTH = 5 | |
| # The node's representative image almost always wins. Contact sheets of the | |
| # alternatives showed why: for Porifera the representative image is a vase | |
| # sponge, while every other free candidate in the clade is an obscure relative. | |
| # Shape only overrules it past exp(1.4), roughly 4:1, where a silhouette is a | |
| # sliver whatever the renderer does. | |
| PRIMARY_BONUS = 1.4 | |
| def get(url, headers=HEADERS, retries=3): | |
| for attempt in range(retries): | |
| try: | |
| with urllib.request.urlopen(urllib.request.Request(url, headers=headers), timeout=60) as response: | |
| return response.read() | |
| except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError) as exc: | |
| if isinstance(exc, urllib.error.HTTPError) and exc.code == 404: | |
| return None | |
| if attempt == retries - 1: | |
| raise | |
| time.sleep(2 * (attempt + 1)) | |
| def candidate(image): | |
| """uuid, licence and aspect ratio of an embedded image object.""" | |
| links = image.get("_links") or {} | |
| sizes = (links.get("rasterFiles") or [{}])[0].get("sizes", "") | |
| uuid = image.get("uuid") or links.get("self", {}).get("href", "").removeprefix("/images/").partition("?")[0] | |
| try: | |
| width, height = (int(part) for part in sizes.split("x")) | |
| except ValueError: | |
| return None | |
| if not uuid or not height: | |
| return None | |
| return {"uuid": uuid, "license": links.get("license", {}).get("href", ""), "ratio": width / height} | |
| def shape_score(found): | |
| """Distance from square, symmetric in either direction.""" | |
| return abs(math.log(found["ratio"])) | |
| def is_free(license_url): | |
| """CC0, Public Domain Mark and plain CC BY only; no NonCommercial, no ShareAlike.""" | |
| return bool(license_url) and "-nc" not in license_url and "-sa" not in license_url | |
| def get_json(url): | |
| body = get(url) | |
| return json.loads(body) if body else None | |
| def visible_taxa(conn, minimum, shallow=SHALLOW_DEPTH): | |
| """Taxa reachable through the columns the atlas draws, largest first. | |
| Two reasons to fetch an icon. Large groups get one because they are where | |
| most assemblies are. The first few columns get one whatever their size, | |
| because a group that borrows an ancestor's silhouette looks identical to its | |
| siblings, and Ichthyosporea, Filasterea, Choanoflagellata and Rotosphaerida | |
| all sit on the opening screen with single-digit assembly counts. | |
| """ | |
| rows = {r["taxid"]: dict(r) for r in conn.execute("SELECT taxid, parent_id, total_count FROM taxa")} | |
| children = {} | |
| for row in rows.values(): | |
| if row["parent_id"] != row["taxid"]: | |
| children.setdefault(row["parent_id"], []).append(row) | |
| for group in children.values(): | |
| group.sort(key=lambda r: (-r["total_count"], r["taxid"])) | |
| seen, stack = set(), [EUKARYOTA] | |
| while stack: | |
| taxid = stack.pop() | |
| if taxid in seen: | |
| continue | |
| seen.add(taxid) | |
| stack.extend(child["taxid"] for child in children.get(taxid, [])[:COLUMN_LIMIT]) | |
| near, level = {EUKARYOTA}, [EUKARYOTA] | |
| for _ in range(shallow): | |
| level = [c["taxid"] for t in level for c in children.get(t, [])[:COLUMN_LIMIT] if c["taxid"] not in near] | |
| near.update(level) | |
| chosen = [t for t in seen if rows[t]["total_count"] >= minimum or t in near] | |
| return sorted(chosen, key=lambda t: -rows[t]["total_count"]) | |
| def build(database, icon_dir, output, minimum=MIN_ASSEMBLIES, pause=0.1): | |
| with sqlite3.connect(f"{Path(database).resolve().as_uri()}?mode=ro", uri=True) as conn: | |
| conn.row_factory = sqlite3.Row | |
| taxa = visible_taxa(conn, minimum) | |
| # Listing endpoints reject a request that does not pin the catalogue build. | |
| build_number = get_json(f"{API}/")["build"] | |
| print(f"Resolving {len(taxa):,} taxa with at least {minimum} assemblies against build {build_number}", flush=True) | |
| nodes, shapes, images, assignments = {}, {}, {}, {} | |
| counts = Counter() | |
| for position, taxid in enumerate(taxa, 1): | |
| resolved = get_json(f"{API}/resolve/ncbi.nlm.nih.gov/taxid/{taxid}?embed_primaryImage=true") | |
| node = (resolved or {}).get("_links", {}).get("self", {}).get("href", "") | |
| node = node.removeprefix("/nodes/").partition("?")[0] | |
| if not node: | |
| counts["unresolved"] += 1 | |
| continue | |
| if node not in nodes: | |
| # Two things decide the pick. The node's own primary image is the one | |
| # PhyloPic considers representative, so it gets a head start. But | |
| # silhouettes range from 0.2 to 9.5 in aspect ratio, and anything far | |
| # from square collapses into a sliver at icon size, so proportion is | |
| # scored too and a badly shaped representative loses to a square one. | |
| primary = (resolved.get("_embedded") or {}).get("primaryImage") or {} | |
| candidates = [] | |
| if primary: | |
| candidates.append((candidate(primary), True)) | |
| listing = get_json(f"{API}/images?build={build_number}&page=0&filter_clade={node}" | |
| f"&filter_license_nc=false&filter_license_sa=false&embed_items=true") | |
| candidates += [(candidate(item), False) for item in (listing or {}).get("_embedded", {}).get("items") or []] | |
| usable = [(shape_score(found) - (PRIMARY_BONUS if is_primary else 0), found, is_primary) | |
| for found, is_primary in candidates if found and is_free(found["license"])] | |
| nodes[node] = "" | |
| if usable: | |
| _, best, is_primary = min(usable, key=lambda entry: entry[0]) | |
| nodes[node] = best["uuid"] | |
| shapes[best["uuid"]] = best["ratio"] | |
| counts["primary_image" if is_primary else "clade_fallback"] += 1 | |
| time.sleep(pause) | |
| image = OVERRIDES.get(taxid) or nodes[node] | |
| if not image: | |
| counts["no_free_image"] += 1 | |
| continue | |
| if taxid in OVERRIDES: | |
| counts["overridden"] += 1 | |
| assignments[taxid] = image | |
| counts["assigned"] += 1 | |
| if position % 100 == 0: | |
| print(f" {position:,}/{len(taxa):,} · {len(set(assignments.values())):,} distinct icons", flush=True) | |
| time.sleep(pause) | |
| icon_dir = Path(icon_dir) | |
| icon_dir.mkdir(parents=True, exist_ok=True) | |
| keep = set() | |
| for image in sorted(set(assignments.values())): | |
| detail = get_json(f"{API}/images/{image}?build={build_number}") | |
| links = (detail or {}).get("_links", {}) | |
| vector = links.get("vectorFile", {}).get("href") | |
| if not vector: | |
| counts["no_vector"] += 1 | |
| continue | |
| target = icon_dir / f"{image}.svg" | |
| if not target.exists(): | |
| body = get(vector, headers={"User-Agent": HEADERS["User-Agent"]}) | |
| if not body: | |
| counts["download_failed"] += 1 | |
| continue | |
| target.write_bytes(body) | |
| time.sleep(pause) | |
| keep.add(image) | |
| sizes = (links.get("rasterFiles") or [{}])[0].get("sizes", "") | |
| if image not in shapes and "x" in sizes: | |
| found = [int(part) for part in sizes.split("x")] | |
| shapes[image] = found[0] / found[1] if found[1] else 1.0 | |
| images[image] = {"license": links.get("license", {}).get("href"), | |
| "attribution": detail.get("attribution"), | |
| "contributor": links.get("contributor", {}).get("title"), | |
| "page": f"https://www.phylopic.org/images/{image}", | |
| "ratio": round(shapes.get(image, 1.0), 3), | |
| "bytes": target.stat().st_size} | |
| assignments = {t: i for t, i in assignments.items() if i in keep} | |
| for stale in icon_dir.glob("*.svg"): | |
| if stale.stem not in keep: | |
| stale.unlink() | |
| counts["removed"] += 1 | |
| licenses = Counter(meta["license"] for meta in images.values()) | |
| unfree = sorted(license for license in licenses if not is_free(license)) | |
| if unfree: | |
| raise ValueError(f"Image with unusable licence slipped through: {unfree}") | |
| payload = {"source": "https://www.phylopic.org", "build": build_number, | |
| "api": f"{API}/resolve/ncbi.nlm.nih.gov/taxid/{{taxid}} then /images?filter_clade=…" | |
| "&filter_license_nc=false&filter_license_sa=false", | |
| "note": "Silhouettes are stored exactly as PhyloPic serves them, unmodified.", | |
| "minimum_assemblies": minimum, | |
| "counts": dict(counts, distinct_icons=len(images), bytes=sum(m["bytes"] for m in images.values())), | |
| "licenses": dict(licenses), | |
| "images": dict(sorted(images.items())), | |
| "taxa": {str(t): assignments[t] for t in sorted(assignments)}} | |
| Path(output).write_text(json.dumps(payload, indent=1)) | |
| print(json.dumps({"counts": payload["counts"], "licenses": payload["licenses"]}, indent=2)) | |
| def main(): | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--database", type=Path, default=ROOT / "data/taxonomy.sqlite") | |
| parser.add_argument("--icon-dir", type=Path, default=ROOT / "data/icons") | |
| parser.add_argument("--output", type=Path, default=ROOT / "data/icons.json") | |
| parser.add_argument("--minimum", type=int, default=MIN_ASSEMBLIES) | |
| args = parser.parse_args() | |
| build(args.database, args.icon_dir, args.output, args.minimum) | |
| if __name__ == "__main__": | |
| main() | |