request-moderator / formatters.py
EmmaScharfmann's picture
EmmaScharfmann HF Staff
Prefill fields from structured submissions instead of guessing
3e83297
Raw
History Blame Contribute Delete
5.64 kB
"""Turn an approved submission into a src/data/*.js entry + a text splice.
The site's data files (src/data/organizations.js, models.js, datasets.js,
blogs.js) are hand-formatted flat JS arrays, not JSON — see the real
examples this module's tests are based on. Rather than parsing/rewriting
the whole file as an AST, we do a plain text insertion, since the existing
formatting is simple and consistent: 2-space indent for `{`/`}`, 4-space
indent for fields, `description`/`excerpt` always on their own line.
"""
import json
# Canonical tag list — mirrors src/data/themes.js themeIds.
VALID_TAGS = [
"biology", "chemistry", "physics", "medicine", "mathematics",
"engineering", "earth-science", "astronomy", "genomics",
"biotechnology", "materials-science", "climate", "energy",
"ecology", "conservation", "benchmark", "scientific-reasoning",
]
# type -> (file path relative to repo root, exported array name, required fields)
TARGETS = {
"organization": ("src/data/organizations.js", "organizations",
["id", "name", "link", "tags"]),
"model": ("src/data/models.js", "models",
["id", "slug", "name", "orgId", "type", "description", "tags"]),
"dataset": ("src/data/datasets.js", "datasets",
["id", "slug", "orgId", "type", "description", "tags"]),
# `slug` and `orgId` are legitimately absent for external (non-HF-blog)
# and non-partnership posts — see e.g. the tamarind.bio entries in
# blogs.js, most of which have slug: null and orgId: null.
"blog": ("src/data/blogs.js", "blogs",
["id", "title", "date", "excerpt", "link", "tags"]),
}
def target_file(type_: str) -> str:
return TARGETS[type_][0]
def missing_fields(type_: str, fields: dict) -> list[str]:
_, _, required = TARGETS[type_]
return [f for f in required if not fields.get(f)]
def _js_string(value) -> str:
return json.dumps(value)
def render_entry(type_: str, fields: dict) -> str:
"""Render a single object literal matching the existing file style."""
if type_ == "organization":
lines = [
" {",
f' id: {_js_string(fields["id"])},',
f' name: {_js_string(fields["name"])},',
f' logo: getOrgLogo({_js_string(fields["id"])}),',
" description:",
f' {_js_string(fields.get("description", ""))},',
f' link: {_js_string(fields["link"])},',
f' tags: {json.dumps(fields["tags"])},',
" },",
]
elif type_ == "model":
lines = [
" {",
f' id: {_js_string(fields["id"])},',
f' slug: {_js_string(fields["slug"])},',
f' name: {_js_string(fields["name"])},',
f' orgId: {_js_string(fields["orgId"])},',
f' type: {_js_string(fields["type"])},',
" description:",
f' {_js_string(fields["description"])},',
f' tags: {json.dumps(fields["tags"])},',
" },",
]
elif type_ == "dataset":
lines = [
" {",
f' id: {_js_string(fields["id"])},',
f' slug: {_js_string(fields["slug"])},',
f' orgId: {_js_string(fields["orgId"])},',
f' type: {_js_string(fields["type"])},',
" description:",
f' {_js_string(fields["description"])},',
f' tags: {json.dumps(fields["tags"])},',
" },",
]
elif type_ == "blog":
lines = [
" {",
f' id: {_js_string(fields["id"])},',
f' title: {_js_string(fields["title"])},',
f' slug: {_js_string(fields["slug"]) if fields.get("slug") else "null"},',
f' orgId: {_js_string(fields["orgId"]) if fields.get("orgId") else "null"},',
f' date: {_js_string(fields["date"])},',
" excerpt:",
f' {_js_string(fields["excerpt"])},',
f' link: {_js_string(fields["link"])},',
f' tags: {json.dumps(fields["tags"])},',
f' featured: {"true" if fields.get("featured") else "false"},',
]
if fields.get("upvotes") is not None:
lines.append(f' upvotes: {int(fields["upvotes"])},')
lines.append(" },")
else:
raise ValueError(f"Unknown submission type: {type_}")
return "\n".join(lines)
def insert_entry(file_content: str, type_: str, entry_block: str, org_id: str | None) -> str:
"""Insert entry_block into file_content, grouped near org_id's other
entries when found, otherwise right after the array's opening line."""
_, array_name, _ = TARGETS[type_]
lines = file_content.split("\n")
opening = f"export const {array_name} = ["
insert_at = None
if org_id:
for i, line in enumerate(lines):
if f'orgId: "{org_id}"' in line:
# walk forward to this object's closing " },"
for j in range(i + 1, len(lines)):
if lines[j] == " },":
insert_at = j + 1
break
# keep scanning in case a later group is a better (last) match
if insert_at is None:
for i, line in enumerate(lines):
if line.strip() == opening:
insert_at = i + 1
break
if insert_at is None:
raise ValueError(f"Could not find `{opening}` in target file")
new_lines = lines[:insert_at] + entry_block.split("\n") + lines[insert_at:]
return "\n".join(new_lines)