tejasashinde's picture
Integrate HF Bucket storage
4784470
Raw
History Blame Contribute Delete
96.5 kB
from __future__ import annotations
import hashlib
import json
import logging
import os
import re
import time
from pathlib import Path
from typing import Iterable
import numpy as np
from huggingface_hub import HfApi, hf_hub_download
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
from backend.cache import load_sample_spaces, load_spaces_cache
from backend.llm import get_llm_service
from backend.storage import get_space_metrics, init_db, upsert_space_metrics
logger = logging.getLogger(__name__)
ORG_NAME = os.getenv("BSQF_ORG_NAME", "build-small-hackathon")
CACHE_TTL = int(os.getenv("BSQF_CACHE_TTL", "600"))
LIVE_METADATA_TTL = int(os.getenv("BSQF_LIVE_METADATA_TTL", "300"))
ROOT_DIR = Path(__file__).resolve().parents[1]
DATA_DIR = ROOT_DIR / "data"
INDEX_PATH = DATA_DIR / "spaces_index.json"
LEGACY_CACHE_PATH = DATA_DIR / "spaces_cache.json"
ORG_READMES_DIR = DATA_DIR / "org_readmes"
ORG_READMES_MANIFEST_PATH = DATA_DIR / "org_readmes_manifest.json"
_STORAGE_READY = False
NOISE_TAGS = {
"gradio",
"streamlit",
"docker",
"static",
"transformers",
"diffusers",
"sentence-transformers",
"pytorch",
"tensorflow",
"spaces",
"huggingface",
"llama-cpp",
"llama.cpp",
"zerogpu",
"nvidia",
"openai",
"anthropic",
"groq",
"cohere",
"gemini",
"qwen",
"llama",
"mistral",
"nemotron",
"minicpm",
"gguf",
"sdk",
"api",
"demo",
}
TRACK_MAP = {
"track:backyard": "Backyard",
"track:backyard-ai": "Backyard",
"track:wood": "Thousand Token Wood",
"track:thousand-token-wood": "Thousand Token Wood",
}
INFRA_REPO_SLUGS = {
"field-guide",
"readme",
"registration",
}
CATEGORY_ORDER = [
"Games, Puzzles & Interactive Fiction",
"Education, Tutors & Learning",
"Writing, Storytelling & Creativity",
"Security, Risk & Privacy",
"Image, Art & Design",
"Voice, Audio, Music & Speech",
"Health, Wellness & Accessibility",
"Search, RAG & Knowledge Tools",
"Developer Tools & Code Assistants",
"Productivity, Agents & Assistants",
"Data, Research & Analysis",
"Community, Social & Collaboration",
"Science, Math & Engineering",
"Other",
]
_CATEGORY_RULES = [
(
"Games, Puzzles & Interactive Fiction",
[
r"\b(game|games|quiz|puzzle|rpg|roguelike|interactive fiction|text adventure|visual novel|wordle|trivia|maze|arcade|card game|board game|mystery|battle|combat|playable)\w*",
],
),
(
"Education, Tutors & Learning",
[
r"\b(educat\w*|learn\w*|tutor\w*|teacher\w*|study\w*|lesson\w*|homework|flashcard\w*|curricul\w*|school\w*|student\w*|explainer\w*|upskill\w*)",
],
),
(
"Writing, Storytelling & Creativity",
[
r"\b(writing|writer(?:s)?|storybook|storytelling|storycraft|story generator|story arc|storyline|poem\w*|poetry|creative\w*|novel\w*|journal\w*|author\w*|script\w*|comic\w*|fanfic\w*|blog(?:ging)?|markdown|branching narrative|interactive fiction|visual novel|bedtime stor(?:y|ies)|picture book)\b",
],
),
(
"Security, Risk & Privacy",
[
r"\b(security|cybersecurity|cyber\s*security|privacy|private|vulnerab\w*|exploit\w*|threat\w*|malware|phishing|scam|fraud|audit\w*|secure\w*|defen\w*|hotfix|cve|advisory|incident response|pentest\w*|infosec|risk path|codebase audit|repository audit)\b",
],
),
(
"Image, Art & Design",
[
r"\b(image\w*|design\w*|logo\w*|poster\w*|illustrat\w*|draw\w*|paint\w*|sketch\w*|art\w*|photo\w*|visual\w*|icon\w*|typograph\w*|gallery|canvas|ui\s*design)\b",
],
),
(
"Voice, Audio, Music & Speech",
[
r"\b(voice\w*|audio\w*|speech\w*|transcrib\w*|tts|asr|podcast\w*|music\w*|song\w*|sound\w*|pronunc\w*|karaoke|dub\w*|caption\w*|sing\w*)",
],
),
(
"Health, Wellness & Accessibility",
[
r"\b(health\w*|medical\w*|clinic\w*|symptom\w*|wellness|accessib\w*|dyslex\w*|mental\w*|therapy\w*|caregiver\w*|nutrition\w*|exercise\w*|rehab\w*|disabilit\w*|a11y)\b",
],
),
(
"Search, RAG & Knowledge Tools",
[
r"\b(search\w*|retriev\w*|rag\b|knowledge base|knowledge graph|index\w*|semantic\w*|embedding\w*|query\w*|qa\b|ask\b|knowledge)\b",
],
),
(
"Developer Tools & Code Assistants",
[
r"\b(dev\w*|code\w*|developer\w*|github|api|cli|terminal|debug\w*|benchmark\w*|deploy\w*|model serving|database|python|javascript|monitoring|logging|testing|assistant for code|copilot)\b",
],
),
(
"Productivity, Agents & Assistants",
[
r"\b(assistant\w*|agent\w*|copilot\w*|productiv\w*|todo\w*|task\w*|calendar\w*|email\w*|note\w*|organi\w*|plan\w*|workflow\w*|automation\w*|second brain|personal ai)\b",
],
),
(
"Data, Research & Analysis",
[
r"\b(data\w*|research\w*|analysis\w*|analytics\w*|dataset\w*|csv|sql|chart\w*|paper\w*|benchmark\w*|experiment\w*|dashboard\w*|report\w*)\b",
],
),
(
"Community, Social & Collaboration",
[
r"\b(community\w*|collaborat\w*|social\w*|forum\w*|chat\w*|discord\w*|feedback\w*|group\w*|team\w*|shared\w*|community|discussion\w*)\b",
],
),
(
"Science, Math & Engineering",
[
r"\b(science\w*|engineering\w*|physics\w*|chemistry\w*|biology\w*|robot\w*|iot\b|mechanic\w*|sensor\w*|lab\w*|simulation\w*|math\w*|signal\w*|satellite\w*|genomics\w*|algebra\w*|calculus\w*|geometry\w*)\b",
],
),
]
_CATEGORY_PATTERNS = [
(name, [re.compile(pattern, re.I) for pattern in patterns]) for name, patterns in _CATEGORY_RULES
]
GENERIC_QUERY_TERMS = {
"app",
"apps",
"application",
"applications",
"related",
"relation",
"spaces",
"space",
"projects",
"project",
"something",
"anything",
"find",
"show",
"suggest",
"suggestion",
"suggestions",
"provide",
"recommend",
"recommendation",
"recommendations",
"looking",
"need",
"want",
"like",
"about",
"around",
"kind",
"good",
"best",
"nice",
"cool",
"great",
"awesome",
"helpful",
"please",
"some",
"any",
"types",
"tools",
"tool",
}
SECURITY_QUERY_HINTS = {
"security",
"cybersecurity",
"cyber",
"privacy",
"private",
"vulnerability",
"vulnerabilities",
"threat",
"threats",
"exploit",
"exploits",
"phishing",
"scam",
"fraud",
"malware",
"audit",
"auditing",
"secure",
"infosec",
"pentest",
"cve",
}
SECURITY_SPACE_HINTS = {
"security",
"cybersecurity",
"privacy",
"private",
"vulnerability",
"vulnerabilities",
"threat",
"threats",
"exploit",
"phishing",
"scam",
"fraud",
"malware",
"audit",
"audits",
"auditing",
"secure",
"advisory",
"incident",
"risk",
"hotfix",
"cve",
"repository",
"codebase",
"attacker",
"attackers",
"defense",
}
CATEGORY_TO_CONCEPT = {
"Games, Puzzles & Interactive Fiction": "games",
"Education, Tutors & Learning": "education",
"Writing, Storytelling & Creativity": "writing",
"Security, Risk & Privacy": "security",
"Image, Art & Design": "image",
"Voice, Audio, Music & Speech": "voice",
"Health, Wellness & Accessibility": "health",
"Search, RAG & Knowledge Tools": "search",
"Developer Tools & Code Assistants": "developer",
"Productivity, Agents & Assistants": "productivity",
"Data, Research & Analysis": "data",
"Community, Social & Collaboration": "community",
"Science, Math & Engineering": "science",
}
INTENT_PRIORITY = [
"kids",
"cybersecurity",
"security",
"games",
"cards",
"writing",
"education",
"voice",
"image",
"search",
"developer",
"productivity",
"data",
"health",
"community",
"science",
]
INTENT_LABELS = {
"kids": "kid-friendly apps",
"cybersecurity": "cybersecurity apps",
"security": "cybersecurity apps",
"games": "games",
"cards": "playing-card games",
"writing": "writing apps",
"education": "learning apps",
"voice": "voice and audio apps",
"image": "image and design apps",
"search": "search and knowledge tools",
"developer": "developer tools",
"productivity": "assistant-style apps",
"data": "data and research tools",
"health": "health and accessibility apps",
"community": "community tools",
"science": "science and engineering apps",
}
CONCEPT_QUERY_TERMS = {
"kids": ["kids", "children", "child", "toddler", "bedtime", "family"],
"cybersecurity": ["cybersecurity", "cyber", "security", "phishing", "scam", "fraud", "malware"],
"security": ["security", "cybersecurity", "privacy", "phishing", "scam", "fraud", "malware"],
"games": ["game", "games", "play", "playable", "puzzle", "quiz", "roguelike"],
"cards": ["card", "cards", "card game", "matching", "memory", "deckbuilder", "poker", "texas hold'em", "tarot"],
"writing": ["writing", "write", "story", "stories", "journal", "blog", "poem", "creative writing"],
"education": ["learn", "learning", "tutor", "study", "teacher", "lesson", "school"],
"voice": ["voice", "audio", "speech", "music", "sound", "tts", "asr"],
"image": ["image", "art", "design", "draw", "illustration", "photo"],
"search": ["search", "rag", "retrieval", "knowledge", "semantic", "embedding"],
"developer": ["developer", "code", "coding", "cli", "terminal", "debug", "deploy"],
"productivity": ["assistant", "agent", "workflow", "task", "notes", "calendar"],
"data": ["data", "research", "analysis", "dashboard", "report"],
"health": ["health", "medical", "wellness", "therapy", "accessibility"],
"community": ["community", "social", "collaboration", "feedback", "chat"],
"science": ["science", "engineering", "math", "physics", "biology", "robotics"],
}
_INTENT_RULES = {
"kids": [
r"\bkid(?:s)?\b",
r"\btoddler(?:s)?\b",
r"\byoung readers?\b",
r"\bbedtime\b",
r"\bstorybook\b",
r"\bfor kids\b",
r"\bchildren'?s\b",
],
"cybersecurity": [
r"\bcybersecurity\b",
r"\bcyber security\b",
r"\bphishing\b",
r"\bscam\b",
r"\bfraud\b",
r"\bmalware\b",
r"\bexploit\w*\b",
r"\bpentest\w*\b",
r"\bcve\b",
r"\binfosec\b",
],
"security": [
r"\bsecurity\b",
r"\bcyber(?:\s*security)?\b",
r"\bprivacy\b",
r"\bphishing\b",
r"\bscam\b",
r"\bfraud\b",
r"\bmalware\b",
r"\bvulnerab\w*\b",
r"\bsecure\b",
r"\binfosec\b",
],
"games": [
r"\bgame(?:s)?\b",
r"\bpuzzle(?:s)?\b",
r"\bquiz(?:zes)?\b",
r"\binteractive fiction\b",
r"\broguelike\b",
r"\bvisual novel\b",
r"\bplay(?:ing|able)?\b",
],
"cards": [
r"\bplaying cards?\b",
r"\bcard game(?:s)?\b",
r"\bcard[- ]matching\b",
r"\bdeckbuilder\b",
r"\bdeck builder\b",
r"\bpoker\b",
r"\btexas hold'?em\b",
r"\btarot deck\b",
r"\b15-card deck\b",
],
"writing": [
r"\bwriting\b",
r"\bjournal(?:ing)?\b",
r"\bblog drafts?\b",
r"\bpoem(?:s)?\b",
r"\bpoetry\b",
r"\bnovel(?:s)?\b",
r"\bcreative writing\b",
r"\bstory generator\b",
r"\bstorybook\b",
r"\bwriting practice\b",
r"\beditable markdown\b",
],
"education": [
r"\blearn(?:ing)?\b",
r"\btutor(?:ing)?\b",
r"\bstudy(?:ing)?\b",
r"\blesson(?:s)?\b",
r"\bschool\b",
r"\bstudent(?:s)?\b",
r"\bteacher\b",
],
"voice": [
r"\bvoice\b",
r"\baudio\b",
r"\bspeech\b",
r"\btts\b",
r"\basr\b",
r"\bmusic\b",
r"\bsound\b",
r"\bkaraoke\b",
],
"image": [
r"\bimage(?:s)?\b",
r"\bart\b",
r"\bdesign\b",
r"\bdraw(?:ing)?\b",
r"\billustrat\w*\b",
r"\bphoto(?:s)?\b",
r"\bvisual\b",
],
"search": [
r"\bsearch\b",
r"\brag\b",
r"\bretriev\w*\b",
r"\bknowledge\b",
r"\bsemantic\b",
r"\bembedding(?:s)?\b",
],
"developer": [
r"\bdeveloper(?:s)?\b",
r"\bcode\b",
r"\bcoding\b",
r"\bcli\b",
r"\bterminal\b",
r"\bdebug\w*\b",
r"\bdeploy\w*\b",
r"\bgithub\b",
],
"productivity": [
r"\bassistant(?:s)?\b",
r"\bagent(?:s)?\b",
r"\bworkflow\b",
r"\btask(?:s)?\b",
r"\bcalendar\b",
r"\bnote(?:s)?\b",
r"\bautomation\b",
],
"data": [
r"\bdata\b",
r"\bresearch\b",
r"\banalysis\b",
r"\banalytics\b",
r"\bdashboard\b",
r"\breport(?:s)?\b",
r"\bdataset(?:s)?\b",
],
"health": [
r"\bhealth\b",
r"\bmedical\b",
r"\bwellness\b",
r"\btherapy\b",
r"\baccessib\w*\b",
r"\bdyslex\w*\b",
],
"community": [
r"\bcommunity\b",
r"\bsocial\b",
r"\bcollaborat\w*\b",
r"\bfeedback\b",
r"\bforum\b",
r"\bchat\b",
],
"science": [
r"\bscience\b",
r"\bengineering\b",
r"\bmath(?:s)?\b",
r"\bphysics\b",
r"\bbiology\b",
r"\brobot\w*\b",
],
}
INTENT_PATTERNS = {
name: [re.compile(pattern, re.I) for pattern in patterns] for name, patterns in _INTENT_RULES.items()
}
_CACHE: list[dict] | None = None
_CACHE_TS: float = 0.0
_LIVE_METADATA_TS: float = 0.0
_INDEX: dict | None = None
_INDEX_TS: float = 0.0
_STATS_LIVE_TS: float = 0.0
_STATS: dict = {
"listed": 0,
"private_skipped": 0,
"infra_skipped": 0,
"readme_fetched": 0,
"readme_missing": 0,
"indexed_pool_size": 0,
"last_refresh": "",
}
def _now_ts() -> float:
return time.time()
def _now_iso() -> str:
from datetime import datetime, timezone
return datetime.now(timezone.utc).isoformat()
def _ensure_parent(path: Path) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
def _normalize(text: str) -> str:
return " ".join((text or "").lower().split())
def _tokenize(text: str) -> list[str]:
return re.findall(r"[a-z0-9][a-z0-9+\-_.]{2,}", _normalize(text))
def _slugify(text: str) -> str:
return re.sub(r"[^a-z0-9]+", "-", (text or "").lower()).strip("-")
def _snapshot_slug(text: str) -> str:
return re.sub(r"[^a-zA-Z0-9._-]+", "__", (text or "").strip()) or "unknown-space"
def _stringify_card_data(card_data: object) -> dict:
if isinstance(card_data, dict):
return dict(card_data)
if hasattr(card_data, "to_dict"):
try:
converted = card_data.to_dict()
if isinstance(converted, dict):
return dict(converted)
except Exception:
return {}
if hasattr(card_data, "__dict__"):
try:
return dict(card_data.__dict__)
except Exception:
return {}
return {}
def _is_private_space(data: dict) -> bool:
return bool(
data.get("private")
or str(data.get("status", "")).strip().lower() == "private"
)
def filter_public_spaces(spaces: list[dict]) -> list[dict]:
return [dict(space) for space in spaces if isinstance(space, dict) and not _is_private_space(space)]
def _extract_space_payload(raw: object) -> dict:
if isinstance(raw, dict):
data = dict(raw)
else:
data = {}
for key in [
"id",
"repo_id",
"name",
"author",
"title",
"summary",
"description",
"tags",
"likes",
"sdk",
"private",
"cardData",
"last_modified",
"lastModified",
"updatedAt",
"disabled",
"gated",
"status",
]:
if hasattr(raw, key):
data[key] = getattr(raw, key)
card_data = _stringify_card_data(data.get("cardData"))
repo_id = str(data.get("id") or data.get("repo_id") or card_data.get("id") or card_data.get("repo_id") or "").strip()
name = str(
data.get("name")
or card_data.get("title")
or data.get("title")
or (repo_id.split("/")[-1].replace("-", " ").title() if repo_id else "Untitled Space")
).strip()
raw_name = str(data.get("name") or repo_id.split("/")[-1] or name).strip()
if "/" in repo_id:
repo_author = repo_id.split("/", 1)[0]
else:
repo_author = ""
author = str(data.get("author") or repo_author).strip()
desc = str(
data.get("description")
or data.get("summary")
or card_data.get("description")
or card_data.get("summary")
or ""
).strip()
tags = data.get("tags") or card_data.get("tags") or []
if isinstance(tags, str):
tags = [tags]
tags = [str(tag).strip() for tag in tags if str(tag).strip()]
likes = int(data.get("likes", card_data.get("likes", 0)) or 0)
sdk = str(data.get("sdk") or card_data.get("sdk") or "unknown").strip()
url = f"https://huggingface.co/spaces/{repo_id}" if repo_id else ""
return {
"id": repo_id,
"name": name,
"raw_name": raw_name,
"author": author,
"url": url,
"tags": tags,
"likes": likes,
"sdk": sdk,
"desc": desc,
"private": bool(data.get("private") or card_data.get("private")),
"disabled": bool(data.get("disabled") or card_data.get("disabled")),
"gated": bool(data.get("gated") or card_data.get("gated")),
"status": str(data.get("status") or card_data.get("status") or "").strip(),
"last_modified": str(data.get("last_modified") or data.get("lastModified") or data.get("updatedAt") or "").strip(),
}
def _is_infra_repo(space_id: str, raw_name: str, name: str, tags: Iterable[str]) -> bool:
slug = _slugify(space_id.rsplit("/", 1)[-1] or raw_name or name)
extra = {
_slugify(item)
for item in os.getenv("BSQF_INFRA_REPOS", "").split(",")
if item.strip()
}
blocked = INFRA_REPO_SLUGS | extra
if slug in blocked:
return True
if any(slug.startswith(prefix) for prefix in ("admin", "infra", "registration", "field-guide")):
return True
lower_tags = {_normalize(tag) for tag in tags}
return bool(lower_tags & {"infra", "admin", "admin:only"})
def _is_hub_private_space(payload: dict) -> bool:
status = str(payload.get("status", "") or "").strip().lower()
return bool(payload.get("private") or status == "private")
def detect_track(tags: list[str]) -> str:
normalized = [_normalize(tag) for tag in tags]
for tag in normalized:
if tag in TRACK_MAP:
return TRACK_MAP[tag]
for tag in normalized:
if "backyard" in tag:
return "Backyard"
if "wood" in tag or "thousand token" in tag:
return "Thousand Token Wood"
return ""
def _useful_tags(tags: list[str]) -> list[str]:
useful: list[str] = []
for tag in tags:
norm = _normalize(tag)
if not norm:
continue
if norm in NOISE_TAGS:
continue
if norm.startswith("track:"):
continue
if norm in useful:
continue
useful.append(norm)
return useful
def _extract_readme_excerpt(readme: str, max_chars: int = 12000) -> str:
cleaned = clean_readme(readme)
if len(cleaned) <= max_chars:
return cleaned
excerpt = cleaned[:max_chars]
last_break = excerpt.rfind("\n\n")
if last_break > 600:
return excerpt[:last_break].strip()
return excerpt.strip()
def fetch_space_readme(space_id: str) -> str:
if not space_id:
return ""
try:
readme_path = hf_hub_download(repo_id=space_id, repo_type="space", filename="README.md")
except Exception:
return ""
try:
text = Path(readme_path).read_text(encoding="utf-8", errors="ignore")
except Exception:
return ""
return text[:20000]
def load_downloaded_space_readme(space_id: str) -> str:
if not space_id:
return ""
readme_path = ORG_READMES_DIR / _snapshot_slug(space_id) / "README.md"
if not readme_path.exists():
return ""
try:
return readme_path.read_text(encoding="utf-8", errors="ignore")[:20000]
except Exception:
return ""
def _load_org_readmes_manifest() -> list[dict]:
if not ORG_READMES_MANIFEST_PATH.exists():
return []
try:
payload = json.loads(ORG_READMES_MANIFEST_PATH.read_text(encoding="utf-8"))
except Exception:
return []
if not isinstance(payload, list):
return []
return [item for item in payload if isinstance(item, dict)]
def _extract_readme_frontmatter(readme: str) -> dict:
if not readme or not readme.lstrip().startswith("---"):
return {"short_description": "", "tags": []}
match = re.match(r"^\s*---\s*\n(.*?)\n---\s*\n?", readme, flags=re.S)
if not match:
return {"short_description": "", "tags": []}
short_description = ""
tags: list[str] = []
in_tags = False
for raw_line in match.group(1).splitlines():
line = raw_line.rstrip()
stripped = line.strip()
if not stripped:
continue
lower = stripped.lower()
if in_tags:
if stripped.startswith("- "):
tag = stripped[2:].strip().strip('"').strip("'")
if tag:
tags.append(tag)
continue
if not line.startswith(" "):
in_tags = False
if lower.startswith("short_description:"):
short_description = stripped.split(":", 1)[1].strip().strip('"').strip("'")
elif lower == "tags:":
in_tags = True
return {"short_description": short_description, "tags": tags}
def _load_manifest_readme(entry: dict, space_id: str) -> str:
readme_text = load_downloaded_space_readme(space_id)
if readme_text.strip():
return readme_text
relative_path = str(entry.get("readme_path", "") or "").strip()
if not relative_path:
return ""
readme_path = ROOT_DIR / relative_path
if not readme_path.exists():
return ""
try:
return readme_path.read_text(encoding="utf-8", errors="ignore")[:20000]
except Exception:
return ""
def _build_manifest_backed_index(existing_spaces: list[dict] | None = None, existing_stats: dict | None = None) -> tuple[list[dict], dict]:
manifest = _load_org_readmes_manifest()
if not manifest:
return [], {}
existing_by_id: dict[str, dict] = {}
for item in existing_spaces or []:
if not isinstance(item, dict):
continue
space_id = str(item.get("id", "")).strip()
if space_id:
existing_by_id[space_id] = dict(item)
for item in _legacy_bootstrap():
if not isinstance(item, dict):
continue
item_dict = dict(item)
space_id = str(item_dict.get("id") or item_dict.get("repo_id") or "").strip()
if not space_id or space_id in existing_by_id:
continue
item_dict.setdefault("id", space_id)
item_dict.setdefault("name", item_dict.get("title", item_dict.get("name", "")))
item_dict.setdefault("raw_name", space_id.split("/", 1)[1] if "/" in space_id else space_id)
item_dict.setdefault("author", space_id.split("/", 1)[0] if "/" in space_id else "")
item_dict.setdefault("url", f"https://huggingface.co/spaces/{space_id}")
item_dict.setdefault("desc", item_dict.get("summary", item_dict.get("desc", "")))
item_dict.setdefault("readme", item_dict.get("readme_text", item_dict.get("readme", "")))
existing_by_id[space_id] = item_dict
records: list[dict] = []
readme_fetched = 0
readme_missing = 0
for entry in manifest:
repo_id = str(entry.get("repo_id", "")).strip()
if not repo_id:
continue
existing = dict(existing_by_id.get(repo_id, {}))
author, _, repo_name = repo_id.partition("/")
readme_text = _load_manifest_readme(entry, repo_id)
frontmatter = _extract_readme_frontmatter(readme_text)
tags = list(dict.fromkeys(list(entry.get("tags", existing.get("tags", [])) or []) + list(frontmatter.get("tags", []) or [])))
if readme_text.strip():
readme_fetched += 1
else:
readme_missing += 1
desc = str(existing.get("desc") or existing.get("summary") or "").strip()
if (not desc or desc == "No summary available yet.") and frontmatter.get("short_description"):
desc = str(frontmatter.get("short_description", "")).strip()
merged = {
"id": repo_id,
"name": str(entry.get("title") or existing.get("name") or repo_name or repo_id),
"raw_name": str(existing.get("raw_name") or repo_name or repo_id),
"author": str(existing.get("author") or author),
"url": str(entry.get("url") or existing.get("url") or f"https://huggingface.co/spaces/{repo_id}"),
"tags": tags,
"likes": int(entry.get("likes", existing.get("likes", 0)) or 0),
"sdk": str(entry.get("sdk") or existing.get("sdk") or "unknown"),
"desc": desc,
"readme": readme_text or str(existing.get("readme") or existing.get("readme_text") or ""),
"readme_summary": str(existing.get("readme_summary", "")),
"track": str(existing.get("track") or detect_track(tags)),
"last_modified": str(existing.get("last_modified", "")),
}
records.append(_normalize_loaded_space(merged))
records.sort(key=lambda item: (-int(item.get("likes", 0) or 0), item.get("name", "").lower()))
prior_stats = existing_stats if isinstance(existing_stats, dict) else {}
stats = {
"listed": len(records),
"private_skipped": None,
"infra_skipped": None,
"readme_fetched": readme_fetched,
"readme_missing": readme_missing,
"indexed_pool_size": len(records),
"last_refresh": _now_iso(),
}
stats["category_distribution"] = _category_distribution_for_spaces(records)
last_live_sync = str(prior_stats.get("last_live_sync", "") or "").strip()
if last_live_sync:
stats["last_live_sync"] = last_live_sync
return records, stats
def hydrate_space_evidence(space: dict, force: bool = False) -> dict:
hydrated = dict(space)
space_id = str(hydrated.get("id", "")).strip()
readme_text = str(hydrated.get("readme", "") or "")
local_readme = load_downloaded_space_readme(space_id)
if local_readme.strip():
readme_text = clean_readme(local_readme)
hydrated["readme"] = readme_text[:20000]
hydrated["readme_summary"] = summarize_readme_for_index(readme_text)[:2500]
hydrated["semantic_profile"] = _build_semantic_profile(
{
"name": str(hydrated.get("name", "")),
"raw_name": str(hydrated.get("raw_name", "")),
"desc": str(hydrated.get("desc", "")),
"tags": list(hydrated.get("tags", []) or []),
"readme_summary": str(hydrated.get("readme_summary", "")),
"category": str(hydrated.get("category", "Other")),
"track": str(hydrated.get("track", "")),
},
readme_text,
)
hydrated["readme_source"] = "local_snapshot"
elif readme_text.strip():
hydrated["readme_source"] = str(hydrated.get("readme_source", "") or "cache")
return hydrated
def clean_readme(text: str) -> str:
text = text or ""
text = re.sub(r"^\s*---\s*\n.*?\n---\s*\n?", "", text, flags=re.S)
text = re.sub(r"!\[[^\]]*\]\([^)]+\)", " ", text)
text = re.sub(r"\[(?:badge|status|build|coverage|license|docs?|demo)[^\]]*\]\([^)]+\)", " ", text, flags=re.I)
text = re.sub(r"<details.*?>.*?</details>\s*", " ", text, flags=re.S | re.I)
text = re.sub(r"<div[^>]*class=['\"]badge[^>]*>.*?</div>\s*", " ", text, flags=re.S | re.I)
text = re.sub(r"^\s*\|.*\|\s*$", " ", text, flags=re.M)
def _code_replacer(match: re.Match) -> str:
body = match.group(1) or ""
if len(body) > 900:
return " "
body = re.sub(r"^\s*#.*$", " ", body, flags=re.M)
return " " + body.strip() + " "
text = re.sub(r"```(?:[a-zA-Z0-9_-]+)?\n(.*?)```", _code_replacer, text, flags=re.S)
text = re.sub(r"^\s*>\s?.*$", " ", text, flags=re.M)
text = re.sub(r"^\s*[-*_]{3,}\s*$", " ", text, flags=re.M)
text = re.sub(r"<script.*?>.*?</script>", " ", text, flags=re.S | re.I)
text = re.sub(r"<style.*?>.*?</style>", " ", text, flags=re.S | re.I)
text = re.sub(r"<[^>]+>", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r"\s*\n\s*", "\n", text)
text = text.strip()
return text
def summarize_readme_for_index(readme: str) -> str:
cleaned = clean_readme(readme)
if not cleaned:
return ""
sections = _readme_sections(cleaned)
if not sections:
return cleaned[:2500]
chosen: list[str] = []
used = set()
section_order = [
"about",
"overview",
"description",
"features",
"what it does",
"how it works",
"how to use",
"usage",
"quickstart",
"getting started",
"examples",
"example",
"model",
"dataset",
"inputs",
"outputs",
]
for needle in section_order:
for title, body in sections:
if title in used:
continue
if needle in _normalize(title):
combined = f"{title}\n{body}".strip()
if combined:
chosen.append(combined)
else:
chosen.append(title)
used.add(title)
break
for title, body in sections:
if title in used:
continue
combined = f"{title}\n{body}".strip()
if not combined:
continue
chosen.append(combined)
used.add(title)
if len(" ".join(chosen)) >= 2300:
break
summary = "\n\n".join(part for part in chosen if part)
if not summary:
summary = _extract_readme_excerpt(cleaned, max_chars=2500)
return summary[:2500]
def _readme_sections(cleaned: str) -> list[tuple[str, str]]:
headings = list(re.finditer(r"^(#{1,6})\s+(.*)$", cleaned, flags=re.M))
if not headings:
cleaned = cleaned.strip()
return [("intro", cleaned)] if cleaned else []
sections: list[tuple[str, str]] = []
intro = cleaned[: headings[0].start()].strip()
if intro:
sections.append(("intro", intro))
for idx, match in enumerate(headings):
start = match.end()
end = headings[idx + 1].start() if idx + 1 < len(headings) else len(cleaned)
title = match.group(2).strip()
body = cleaned[start:end].strip()
sections.append((title, body))
return sections
def _split_semantic_lines(text: str) -> list[str]:
if not text:
return []
pieces = []
for part in re.split(r"(?<=[.!?])\s+|\n+", text):
cleaned = re.sub(r"^\s*(?:[-*]|\d+[.)])\s+", "", part).strip()
cleaned = re.sub(r"\s+", " ", cleaned).strip(" .:-")
if cleaned:
pieces.append(cleaned)
return pieces
def _extract_audience_phrase(text: str) -> str:
if not text:
return ""
patterns = [
r"\bfor\s+([^.:\n]{4,80})",
r"\bbuilt for\s+([^.:\n]{4,80})",
r"\bdesigned for\s+([^.:\n]{4,80})",
r"\bused by\s+([^.:\n]{4,80})",
r"\bhelps\s+([^.:\n]{4,80})",
]
lowered = _normalize(text)
for pattern in patterns:
match = re.search(pattern, lowered, flags=re.I)
if match:
phrase = match.group(1).strip(" .:-")
if phrase and len(phrase.split()) >= 2:
return phrase
return ""
def _semantic_profile_text(profile: dict) -> str:
parts = [
str(profile.get("primary_activity", "")),
str(profile.get("audience", "")),
" ".join(profile.get("core_actions", []) or []),
" ".join(profile.get("value_points", []) or []),
" ".join(profile.get("key_snippets", []) or []),
" ".join(profile.get("concepts", []) or []),
]
return " ".join(part for part in parts if part).strip()
def _build_semantic_profile(space: dict, readme_text: str | None = None) -> dict:
title = str(space.get("name", "")).strip()
raw_name = str(space.get("raw_name", "")).strip()
desc = str(space.get("desc", "")).strip()
readme_summary = str(space.get("readme_summary", "")).strip()
category = str(space.get("category", "Other")).strip()
track = str(space.get("track", "")).strip()
tags = [str(tag).strip() for tag in list(space.get("tags", []) or []) if str(tag).strip()]
clean_readme_text = clean_readme(readme_text if readme_text is not None else str(space.get("readme", "") or ""))
sections = _readme_sections(clean_readme_text)
section_text = "\n\n".join(body for _, body in sections if body)
semantic_source = "\n".join(part for part in [title, raw_name, desc, readme_summary, section_text] if part)
identity_terms = _query_terms(" ".join([title, raw_name, desc, " ".join(tags), readme_summary, category, track]))
snippets: list[str] = []
for source in [desc, readme_summary, section_text, clean_readme_text]:
if not source:
continue
snippet = _extract_evidence_snippet(source, identity_terms)
cleaned = re.sub(r"[#*_`>\[\]]", " ", snippet)
cleaned = re.sub(r"\s+", " ", cleaned).strip(" .:-")
if cleaned and cleaned not in snippets:
snippets.append(cleaned[:220])
if len(snippets) >= 4:
break
primary_activity = _infer_space_mechanic(
{
"name": title,
"raw_name": raw_name,
"desc": desc,
"readme_summary": readme_summary,
},
semantic_source or desc or readme_summary,
)
audience = _extract_audience_phrase(semantic_source)
value_points: list[str] = []
for snippet in snippets:
lowered = snippet.lower()
if any(marker in lowered for marker in (" helps ", " lets ", " allows ", " keeps ", " generates ", " shows ", " returns ", " flags ", " turns ", " provides ", " surfaces ", " guides ")):
value_points.append(snippet)
if not value_points and snippets:
value_points.append(snippets[0])
concept_basis = "\n".join(
[
title,
raw_name,
desc,
" ".join(tags),
readme_summary,
category,
track,
section_text[:2000],
]
)
concepts = _detect_intent_concepts(concept_basis)
category_concept = CATEGORY_TO_CONCEPT.get(category)
if category_concept:
concepts.add(category_concept)
confidence = 0.35
if desc:
confidence += 0.1
if readme_summary:
confidence += 0.1
if snippets:
confidence += 0.15
if audience:
confidence += 0.05
if category and category != "Other":
confidence += 0.05
confidence += min(len(concepts), 4) * 0.03
confidence = round(min(confidence, 0.95), 2)
profile = {
"primary_activity": primary_activity,
"audience": audience,
"core_actions": snippets[:4],
"value_points": value_points[:3],
"key_snippets": snippets[:4],
"concepts": sorted(concepts),
"confidence": confidence,
"source": "local_snapshot" if clean_readme_text else "cache",
}
profile_text = _semantic_profile_text(profile)
if profile_text:
profile["profile_text"] = profile_text
return profile
def categorize_space(
title: str,
raw_name: str,
desc: str,
tags: list[str],
readme_summary: str,
) -> str:
text_parts = {
"title": _normalize(title),
"raw_name": _normalize(raw_name),
"desc": _normalize(desc),
"tags": _normalize(" ".join(_useful_tags(tags))),
"readme_summary": _normalize(readme_summary),
}
field_weights = {
"title": 5.0,
"raw_name": 4.0,
"desc": 3.0,
"tags": 4.0,
"readme_summary": 1.75,
}
best_category = "Other"
best_score = 0.0
for category, patterns in _CATEGORY_PATTERNS:
score = 0.0
for field_name, text in text_parts.items():
if not text:
continue
for pattern in patterns:
match_count = len(pattern.findall(text))
if match_count:
score += field_weights[field_name] * match_count
if category == "Security, Risk & Privacy":
score *= 1.2
if score > best_score:
best_category = category
best_score = score
if best_score >= 2.5:
return best_category
return "Other"
def _build_search_text(name: str, raw_name: str, desc: str, useful_tags: list[str], readme_summary: str, readme: str) -> str:
return _normalize(
"\n".join(
[
name,
raw_name,
desc,
" ".join(useful_tags),
readme_summary,
readme,
]
)
)
def _detect_intent_concepts(text: str) -> set[str]:
normalized = _normalize(text)
if not normalized:
return set()
concepts = {
name
for name, patterns in INTENT_PATTERNS.items()
if any(pattern.search(normalized) for pattern in patterns)
}
return concepts
def _space_concept_text(space: dict) -> str:
return "\n".join(
[
str(space.get("name", "")),
str(space.get("raw_name", "")),
str(space.get("desc", "")),
" ".join(space.get("tags", []) or []),
str(space.get("readme_summary", "")),
str(space.get("category", "")),
]
)
def _space_concepts(space: dict) -> set[str]:
existing = space.get("concepts", []) or []
concepts = {str(item).strip() for item in existing if str(item).strip()}
if not concepts:
concepts = _detect_intent_concepts(_space_concept_text(space))
category_concept = CATEGORY_TO_CONCEPT.get(str(space.get("category", "")), "")
if category_concept:
concepts.add(category_concept)
return concepts
def _build_intent_profile(user_query: str, liked_text: str = "") -> dict:
query_concepts = _detect_intent_concepts(user_query)
liked_concepts = _detect_intent_concepts(liked_text)
if re.search(r"\bchild(?:ren)?\b", _normalize(user_query)):
query_concepts.add("kids")
if re.search(r"\bchild(?:ren)?\b", _normalize(liked_text)):
liked_concepts.add("kids")
if "cards" in query_concepts:
query_concepts.add("games")
if "cards" in liked_concepts:
liked_concepts.add("games")
if "cybersecurity" in query_concepts:
query_concepts.discard("security")
if "cybersecurity" in liked_concepts:
liked_concepts.discard("security")
required_concepts = set(query_concepts)
preferred_concepts = liked_concepts - required_concepts
required_categories = {
category for category, concept in CATEGORY_TO_CONCEPT.items() if concept in required_concepts
}
if "cybersecurity" in required_concepts:
required_categories.add("Security, Risk & Privacy")
return {
"required_concepts": required_concepts,
"preferred_concepts": preferred_concepts,
"required_categories": required_categories,
}
def _ordered_concepts(concepts: Iterable[str]) -> list[str]:
concept_set = {item for item in concepts if item}
ordered = [name for name in INTENT_PRIORITY if name in concept_set]
ordered.extend(sorted(item for item in concept_set if item not in ordered))
return ordered
def _compose_intent_summary(profile: dict, fallback_query: str = "") -> str:
required = set(profile.get("required_concepts", set()) or set())
normalized_query = _normalize(fallback_query)
if {"kids", "education"} <= required and "tutor" in normalized_query:
return "kid-friendly tutor apps"
if "education" in required and "tutor" in normalized_query:
return "tutor apps"
if {"kids", "security"} <= required:
return "kid-friendly cybersecurity apps"
if {"games", "cards"} <= required:
return "playing-card games"
if {"kids", "writing"} <= required:
return "kid-friendly writing apps"
if {"kids", "education"} <= required:
return "kid-friendly learning apps"
ordered = _ordered_concepts(required)
labels = [INTENT_LABELS.get(name, name) for name in ordered[:3]]
if not labels:
return (fallback_query or "the request").strip()
if len(labels) == 1:
return labels[0]
if len(labels) == 2:
return f"{labels[0]} and {labels[1]}"
return f"{labels[0]}, {labels[1]}, and {labels[2]}"
def _query_alignment_terms(user_query: str, liked_text: str, profile: dict) -> list[str]:
terms = _query_terms(user_query) + _query_terms(liked_text)
for concept in _ordered_concepts(
set(profile.get("required_concepts", set()) or set()) | set(profile.get("preferred_concepts", set()) or set())
):
for token in CONCEPT_QUERY_TERMS.get(concept, []):
if token not in terms:
terms.append(token)
deduped: list[str] = []
for term in terms:
if term and term not in deduped:
deduped.append(term)
return deduped[:18]
def _space_intent_alignment(space: dict, profile: dict) -> dict:
space_concepts = _space_concepts(space)
required = set(profile.get("required_concepts", set()) or set())
preferred = set(profile.get("preferred_concepts", set()) or set())
matched_required = required & space_concepts
matched_preferred = preferred & space_concepts
missing_required = required - space_concepts
return {
"space_concepts": space_concepts,
"matched_required": matched_required,
"matched_preferred": matched_preferred,
"missing_required": missing_required,
}
def _query_aligned_evidence_snippets(space: dict, user_query: str, liked_text: str, profile: dict, limit: int = 3) -> list[str]:
terms = _query_alignment_terms(user_query, liked_text, profile)
ranked_snippets: dict[str, int] = {}
sources = [
(str(space.get("desc", "")), 2),
(str(space.get("readme_summary", "")), 1),
(str(space.get("readme", "")), 0),
]
for source, source_bonus in sources:
if not source:
continue
snippet = _extract_evidence_snippet(source, terms)
cleaned = re.sub(r"[#*_`>\[\]]", " ", snippet)
cleaned = re.sub(r"\s+", " ", cleaned).strip(" .:-")
if not cleaned:
continue
trimmed = cleaned[:220]
ranked_snippets[trimmed] = max(
ranked_snippets.get(trimmed, -999),
_evidence_snippet_rank(trimmed, terms) + source_bonus,
)
ordered = sorted(ranked_snippets.items(), key=lambda item: (-item[1], -len(item[0]), item[0].lower()))
return [snippet for snippet, _ in ordered[:limit]]
def _query_terms(text: str) -> list[str]:
terms: list[str] = []
for token in _tokenize(text):
if token in GENERIC_QUERY_TERMS or token in terms:
continue
terms.append(token)
return terms
def _query_domains(query_terms: list[str]) -> set[str]:
domains: set[str] = set()
if any(term in SECURITY_QUERY_HINTS for term in query_terms):
domains.add("security")
return domains
def _expanded_query_terms(query_terms: list[str]) -> list[str]:
expanded = list(query_terms)
domains = _query_domains(query_terms)
if "security" in domains:
for token in [
"security",
"cybersecurity",
"privacy",
"vulnerability",
"threat",
"audit",
"phishing",
"scam",
"malware",
"secure",
"cve",
"risk",
]:
if token not in expanded:
expanded.append(token)
return expanded
def _domain_signal_score(space: dict, domains: set[str]) -> tuple[float, list[str]]:
if not domains:
return 0.0, []
haystack = _normalize(
" ".join(
[
str(space.get("name", "")),
str(space.get("desc", "")),
" ".join(space.get("tags", []) or []),
str(space.get("readme_summary", "")),
str(space.get("readme", ""))[:4000],
str(space.get("category", "")),
]
)
)
signals: list[str] = []
score = 0.0
if "security" in domains:
hits = [term for term in SECURITY_SPACE_HINTS if term in haystack]
if hits:
score += min(8.0, 1.3 * len(hits))
signals.extend(hits[:4])
if str(space.get("category", "")) == "Security, Risk & Privacy":
score += 3.0
deduped_signals: list[str] = []
for signal in signals:
if signal not in deduped_signals:
deduped_signals.append(signal)
return score, deduped_signals
def _legacy_bootstrap() -> list[dict]:
cached = load_spaces_cache()
if cached:
return [space.to_dict() for space in cached]
sample = load_sample_spaces()
if sample:
return [space.to_dict() for space in sample]
return []
def _load_index_file() -> tuple[list[dict], dict]:
if not INDEX_PATH.exists():
return [], {}
try:
payload = json.loads(INDEX_PATH.read_text(encoding="utf-8"))
except Exception:
return [], {}
if isinstance(payload, dict):
spaces = payload.get("spaces", [])
stats = payload.get("stats", {})
if isinstance(spaces, list):
return [item for item in spaces if isinstance(item, dict)], stats if isinstance(stats, dict) else {}
return [], {}
if isinstance(payload, list):
return [item for item in payload if isinstance(item, dict)], {}
return [], {}
def _save_index_file(spaces: list[dict], stats: dict) -> None:
_ensure_parent(INDEX_PATH)
INDEX_PATH.write_text(
json.dumps(
{
"stats": stats,
"spaces": spaces,
},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
def _ensure_storage_ready() -> None:
global _STORAGE_READY
if _STORAGE_READY:
return
init_db()
_STORAGE_READY = True
def _is_fresh_file(path: Path, ttl: int = CACHE_TTL) -> bool:
try:
return path.exists() and (_now_ts() - path.stat().st_mtime) < ttl
except Exception:
return False
def _refresh_spaces_from_hub() -> tuple[list[dict], dict]:
api = HfApi()
try:
raw_spaces = list(api.list_spaces(author=ORG_NAME, full=True))
except TypeError:
try:
raw_spaces = list(api.list_spaces(author=ORG_NAME))
except Exception as exc:
logger.warning("Hub space listing failed for %s: %s", ORG_NAME, exc)
return [], {}
except Exception as exc:
logger.warning("Hub space listing failed for %s: %s", ORG_NAME, exc)
return [], {}
listed = 0
private_skipped = 0
infra_skipped = 0
readme_fetched = 0
readme_missing = 0
records: list[dict] = []
for raw in raw_spaces:
listed += 1
payload = _extract_space_payload(raw)
space_id = payload["id"]
if not space_id:
continue
if _is_hub_private_space(payload):
private_skipped += 1
continue
if _is_infra_repo(space_id, payload["raw_name"], payload["name"], payload["tags"]):
infra_skipped += 1
continue
readme = load_downloaded_space_readme(space_id)
if readme:
readme_fetched += 1
else:
readme_missing += 1
clean = clean_readme(readme)
readme_summary = summarize_readme_for_index(clean)
useful_tags = _useful_tags(payload["tags"])
category = categorize_space(
payload["name"],
payload["raw_name"],
payload["desc"],
useful_tags,
readme_summary,
)
track = detect_track(payload["tags"])
semantic_profile = _build_semantic_profile(
{
"name": payload["name"],
"raw_name": payload["raw_name"],
"desc": payload["desc"],
"tags": useful_tags,
"readme_summary": readme_summary,
"category": category,
"track": track,
},
clean,
)
semantic_profile_text = _semantic_profile_text(semantic_profile)
search_text = _build_search_text(
payload["name"],
payload["raw_name"],
payload["desc"],
useful_tags,
readme_summary,
_extract_readme_excerpt(clean, max_chars=12000),
)
if semantic_profile_text:
search_text = _normalize("\n".join([search_text, semantic_profile_text]))
records.append(
{
"id": space_id,
"name": payload["name"],
"raw_name": payload["raw_name"],
"author": payload["author"],
"url": payload["url"],
"tags": useful_tags,
"likes": payload["likes"],
"sdk": payload["sdk"],
"desc": payload["desc"],
"readme": clean[:20000],
"readme_summary": readme_summary,
"category": category,
"track": track,
"search_text": search_text,
"semantic_profile": semantic_profile,
"concepts": sorted(
_detect_intent_concepts(
"\n".join([payload["name"], payload["desc"], " ".join(useful_tags), readme_summary, category])
)
| ({CATEGORY_TO_CONCEPT[category]} if category in CATEGORY_TO_CONCEPT else set())
),
"cardData": None,
"last_modified": payload["last_modified"],
}
)
records.sort(key=lambda item: (-int(item.get("likes", 0) or 0), item.get("name", "").lower()))
stats = {
"listed": listed,
"private_skipped": private_skipped,
"infra_skipped": infra_skipped,
"readme_fetched": readme_fetched,
"readme_missing": readme_missing,
"indexed_pool_size": len(records),
"last_refresh": _now_iso(),
}
return records, stats
def _store_space_metrics_from_spaces(spaces: list[dict]) -> None:
synced_at = _now_iso()
payloads: list[dict] = []
for space in spaces:
repo_id = str(space.get("id", "")).strip()
if not repo_id:
continue
payloads.append(
{
"repo_id": repo_id,
"likes": int(space.get("likes", 0) or 0),
"synced_at": synced_at,
"updated_at": synced_at,
}
)
upsert_space_metrics(payloads)
def _refresh_live_catalog_stats(force: bool = False) -> bool:
global _CACHE, _CACHE_TS, _LIVE_METADATA_TS, _INDEX, _INDEX_TS, _STATS_LIVE_TS
if not force and _STATS_LIVE_TS and (_now_ts() - _STATS_LIVE_TS) < LIVE_METADATA_TTL:
return False
_ensure_storage_ready()
refreshed, stats = _refresh_spaces_from_hub()
_STATS_LIVE_TS = _now_ts()
if not refreshed:
return False
_store_space_metrics_from_spaces(refreshed)
refreshed, _ = _apply_space_metrics(refreshed)
_CACHE = [_normalize_loaded_space(item) for item in refreshed]
_CACHE_TS = _STATS_LIVE_TS
_LIVE_METADATA_TS = _CACHE_TS
_INDEX = None
_INDEX_TS = 0.0
_STATS.update(stats)
_STATS["listed"] = int(stats.get("listed", len(_CACHE)) or len(_CACHE))
_STATS["private_skipped"] = int(stats.get("private_skipped", 0) or 0)
_STATS["infra_skipped"] = int(stats.get("infra_skipped", 0) or 0)
_STATS["readme_fetched"] = int(stats.get("readme_fetched", 0) or 0)
_STATS["readme_missing"] = int(stats.get("readme_missing", 0) or 0)
_STATS.pop("code_fetched", None)
_STATS.pop("code_missing", None)
_STATS["indexed_pool_size"] = len(_CACHE)
_STATS["category_distribution"] = _category_distribution_for_spaces(_CACHE)
_STATS["last_refresh"] = str(stats.get("last_refresh", "") or _now_iso())
_save_index_file(_CACHE, _STATS)
_build_runtime_index(_CACHE)
return True
def _list_live_space_payloads() -> dict[str, dict]:
api = HfApi()
try:
raw_spaces = list(api.list_spaces(author=ORG_NAME, full=True))
except TypeError:
try:
raw_spaces = list(api.list_spaces(author=ORG_NAME))
except Exception as exc:
logger.warning("Live metadata sync failed for %s: %s", ORG_NAME, exc)
return {}
except Exception as exc:
logger.warning("Live metadata sync failed for %s: %s", ORG_NAME, exc)
return {}
payloads: dict[str, dict] = {}
for raw in raw_spaces:
payload = _extract_space_payload(raw)
space_id = payload["id"]
if not space_id:
continue
if _is_hub_private_space(payload):
continue
if _is_infra_repo(space_id, payload["raw_name"], payload["name"], payload["tags"]):
continue
payloads[space_id] = payload
return payloads
def _normalize_loaded_space(data: dict) -> dict:
space_id = str(data.get("id", ""))
tags = data.get("tags", []) or []
if isinstance(tags, str):
tags = [tags]
local_readme = load_downloaded_space_readme(space_id)
frontmatter = _extract_readme_frontmatter(local_readme)
useful_tags = _useful_tags([str(tag) for tag in list(tags) + list(frontmatter.get("tags", []) or []) if str(tag).strip()])
clean_readme_text = clean_readme(local_readme or str(data.get("readme", "")))
clean_readme_summary = clean_readme(str(data.get("readme_summary", "")))
desc = str(data.get("desc", "") or "").strip()
if (not desc or desc == "No summary available yet.") and frontmatter.get("short_description"):
desc = str(frontmatter.get("short_description", "")).strip()
if not clean_readme_summary:
clean_readme_summary = summarize_readme_for_index(clean_readme_text)
semantic_profile = data.get("semantic_profile")
if not isinstance(semantic_profile, dict) or not semantic_profile:
semantic_profile = _build_semantic_profile(
{
"name": str(data.get("name", "")),
"raw_name": str(data.get("raw_name", "")),
"desc": desc,
"tags": useful_tags,
"readme_summary": clean_readme_summary,
"category": str(data.get("category", "Other")),
"track": str(data.get("track", "")),
},
clean_readme_text,
)
semantic_profile_text = _semantic_profile_text(semantic_profile)
search_text = data.get("search_text") or _build_search_text(
str(data.get("name", "")),
str(data.get("raw_name", "")),
desc,
useful_tags,
clean_readme_summary,
_extract_readme_excerpt(clean_readme_text, max_chars=12000),
)
if semantic_profile_text:
search_text = _normalize("\n".join([str(search_text), semantic_profile_text]))
category = categorize_space(
str(data.get("name", "")),
str(data.get("raw_name", "")),
str(data.get("desc", "")),
list(data.get("tags", []) or []),
clean_readme_summary,
)
if category not in CATEGORY_ORDER:
category = "Other"
concepts = _detect_intent_concepts(
"\n".join(
[
str(data.get("name", "")),
str(data.get("raw_name", "")),
str(data.get("desc", "")),
" ".join(useful_tags),
clean_readme_summary,
category,
]
)
)
category_concept = CATEGORY_TO_CONCEPT.get(category)
if category_concept:
concepts.add(category_concept)
return {
"id": space_id,
"name": str(data.get("name", "")),
"raw_name": str(data.get("raw_name", "")),
"author": str(data.get("author", "")),
"url": str(data.get("url", "")),
"tags": useful_tags,
"likes": int(data.get("likes", 0) or 0),
"sdk": str(data.get("sdk", "unknown")),
"desc": desc,
"readme": clean_readme_text[:20000],
"readme_summary": clean_readme_summary[:2500],
"category": category,
"track": str(data.get("track", "")),
"search_text": _normalize(search_text),
"semantic_profile": semantic_profile,
"concepts": sorted(concepts),
"last_modified": str(data.get("last_modified", "")),
}
def _category_distribution_for_spaces(spaces: list[dict]) -> dict[str, int]:
counts: dict[str, int] = {}
for space in spaces:
category = str(space.get("category") or "Other")
counts[category] = counts.get(category, 0) + 1
return dict(sorted(counts.items(), key=lambda item: (-item[1], item[0].lower())))
def _seed_missing_space_metrics(spaces: list[dict]) -> bool:
repo_ids = [str(space.get("id", "")).strip() for space in spaces if str(space.get("id", "")).strip()]
if not repo_ids:
return False
existing = get_space_metrics(repo_ids)
synced_at = _now_iso()
missing_rows: list[dict] = []
for space in spaces:
repo_id = str(space.get("id", "")).strip()
if not repo_id or repo_id in existing:
continue
missing_rows.append(
{
"repo_id": repo_id,
"likes": int(space.get("likes", 0) or 0),
"synced_at": synced_at,
"updated_at": synced_at,
}
)
if not missing_rows:
return False
upsert_space_metrics(missing_rows)
return True
def _apply_space_metrics(spaces: list[dict]) -> tuple[list[dict], bool]:
repo_ids = [str(space.get("id", "")).strip() for space in spaces if str(space.get("id", "")).strip()]
if not repo_ids:
return [dict(space) for space in spaces], False
metrics = get_space_metrics(repo_ids)
if not metrics:
return [dict(space) for space in spaces], False
changed = False
merged_spaces: list[dict] = []
for space in spaces:
merged = dict(space)
repo_id = str(merged.get("id", "")).strip()
metric = metrics.get(repo_id)
if metric:
likes = int(metric.get("likes", merged.get("likes", 0)) or 0)
if int(merged.get("likes", 0) or 0) != likes:
changed = True
merged["likes"] = likes
if metric.get("synced_at"):
merged["likes_synced_at"] = str(metric.get("synced_at", "") or "")
merged_spaces.append(merged)
return merged_spaces, changed
def _load_from_disk_or_legacy() -> list[dict]:
spaces, stats = _load_index_file()
manifest = _load_org_readmes_manifest()
if manifest:
expected_listed = len([item for item in manifest if str(item.get("repo_id", "")).strip()])
expected_fetched = 0
expected_missing = 0
for entry in manifest:
repo_id = str(entry.get("repo_id", "")).strip()
if not repo_id:
continue
if _load_manifest_readme(entry, repo_id).strip():
expected_fetched += 1
else:
expected_missing += 1
should_rebuild_from_manifest = (
len(spaces) != expected_listed
or int(stats.get("indexed_pool_size", 0) or 0) != expected_listed
or int(stats.get("readme_fetched", 0) or 0) != expected_fetched
or int(stats.get("readme_missing", 0) or 0) != expected_missing
)
if should_rebuild_from_manifest:
rebuilt_spaces, rebuilt_stats = _build_manifest_backed_index(spaces, stats)
if rebuilt_spaces:
_STATS.update(rebuilt_stats)
_save_index_file(rebuilt_spaces, rebuilt_stats)
return rebuilt_spaces
if spaces:
normalized = [_normalize_loaded_space(item) for item in spaces]
_STATS.update(stats or {})
_STATS.pop("code_fetched", None)
_STATS.pop("code_missing", None)
_STATS["listed"] = int(_STATS.get("listed", len(normalized)) or len(normalized))
if _STATS.get("private_skipped") is None and "private_skipped" in stats:
_STATS["private_skipped"] = stats.get("private_skipped")
if _STATS.get("infra_skipped") is None and "infra_skipped" in stats:
_STATS["infra_skipped"] = stats.get("infra_skipped")
_STATS["indexed_pool_size"] = len(normalized)
_STATS["category_distribution"] = _category_distribution_for_spaces(normalized)
if not _STATS.get("last_refresh"):
_STATS["last_refresh"] = _now_iso()
_save_index_file(normalized, _STATS)
return normalized
legacy = _legacy_bootstrap()
if legacy:
normalized = []
for item in legacy:
item = dict(item)
item.setdefault("id", item.get("repo_id", ""))
item.setdefault("name", item.get("title", item.get("name", "")))
item.setdefault("raw_name", item.get("repo_id", item.get("raw_name", "")))
item.setdefault("author", item.get("repo_id", "").split("/")[0] if "/" in str(item.get("repo_id", "")) else "")
item.setdefault("url", item.get("url", f"https://huggingface.co/spaces/{item.get('repo_id', '')}"))
item.setdefault("desc", item.get("summary", item.get("desc", "")))
item.setdefault("readme", item.get("readme_text", item.get("readme", "")))
item.setdefault("readme_summary", summarize_readme_for_index(str(item.get("readme", ""))))
item.setdefault("category", categorize_space(
str(item.get("title", item.get("name", ""))),
str(item.get("raw_name", "")),
str(item.get("desc", "")),
list(item.get("tags", []) or []),
str(item.get("readme_summary", "")),
))
item.setdefault("track", detect_track(list(item.get("tags", []) or [])))
item["search_text"] = _normalize(
_build_search_text(
str(item.get("name", "")),
str(item.get("raw_name", "")),
str(item.get("desc", "")),
_useful_tags(list(item.get("tags", []) or [])),
str(item.get("readme_summary", "")),
str(item.get("readme", "")),
)
)
normalized.append(_normalize_loaded_space(item))
_STATS["listed"] = int(_STATS.get("listed", len(normalized)) or len(normalized))
_STATS["indexed_pool_size"] = len(normalized)
_STATS["category_distribution"] = _category_distribution_for_spaces(normalized)
return normalized
return []
def load_spaces(force: bool = False) -> list[dict]:
global _CACHE, _CACHE_TS, _INDEX, _INDEX_TS, _LIVE_METADATA_TS
_ensure_storage_ready()
if not force and _CACHE and (_now_ts() - _CACHE_TS) < CACHE_TTL:
return [dict(item) for item in _CACHE]
loaded = _load_from_disk_or_legacy()
if loaded:
seeded = _seed_missing_space_metrics(loaded)
loaded, metrics_changed = _apply_space_metrics(loaded)
_CACHE = filter_public_spaces(loaded)
_CACHE_TS = _now_ts()
_STATS["indexed_pool_size"] = len(_CACHE)
if not _STATS.get("listed"):
_STATS["listed"] = len(_CACHE)
if not _STATS.get("last_refresh"):
_STATS["last_refresh"] = _now_iso()
if seeded or metrics_changed or not INDEX_PATH.exists():
_save_index_file(_CACHE, _STATS)
_build_runtime_index(_CACHE)
return [dict(item) for item in _CACHE]
if not force:
_CACHE = []
_CACHE_TS = _now_ts()
_INDEX = None
_INDEX_TS = 0.0
return []
refreshed, stats = _refresh_spaces_from_hub()
if refreshed:
_store_space_metrics_from_spaces(refreshed)
refreshed, _ = _apply_space_metrics(refreshed)
_CACHE = [_normalize_loaded_space(item) for item in refreshed]
_CACHE_TS = _now_ts()
_LIVE_METADATA_TS = _CACHE_TS
_STATS.update(stats)
_STATS["private_skipped"] = int(stats.get("private_skipped", 0) or 0)
_STATS["infra_skipped"] = int(stats.get("infra_skipped", 0) or 0)
_STATS["indexed_pool_size"] = len(_CACHE)
_STATS["category_distribution"] = _category_distribution_for_spaces(_CACHE)
_save_index_file(_CACHE, _STATS)
_build_runtime_index(_CACHE)
return [dict(item) for item in _CACHE]
_CACHE = []
_CACHE_TS = _now_ts()
_INDEX = None
_INDEX_TS = 0.0
return []
def refresh_space_metrics(force: bool = False) -> bool:
global _CACHE, _CACHE_TS, _LIVE_METADATA_TS
if not force and _LIVE_METADATA_TS and (_now_ts() - _LIVE_METADATA_TS) < LIVE_METADATA_TTL:
return False
_ensure_storage_ready()
current_spaces = [dict(item) for item in (_CACHE or _load_from_disk_or_legacy())]
if not current_spaces:
return False
live_payloads = _list_live_space_payloads()
if not live_payloads:
return False
changed = False
updated_spaces: list[dict] = []
metrics_payloads: list[dict] = []
synced_at = _now_iso()
for item in current_spaces:
merged = dict(item)
live = live_payloads.get(str(item.get("id", "")).strip())
if live:
live_likes = int(live.get("likes", merged.get("likes", 0)) or 0)
if int(merged.get("likes", 0) or 0) != live_likes:
merged["likes"] = live_likes
changed = True
metrics_payloads.append(
{
"repo_id": str(item.get("id", "")).strip(),
"likes": live_likes,
"synced_at": synced_at,
"updated_at": synced_at,
}
)
updated_spaces.append(_normalize_loaded_space(merged))
if metrics_payloads:
upsert_space_metrics(metrics_payloads)
updated_spaces.sort(key=lambda row: (-int(row.get("likes", 0) or 0), row.get("name", "").lower()))
_CACHE = updated_spaces
_CACHE_TS = _now_ts()
_LIVE_METADATA_TS = _CACHE_TS
_STATS["indexed_pool_size"] = len(updated_spaces)
_STATS["last_live_sync"] = _now_iso()
_save_index_file(updated_spaces, _STATS)
return changed
def _build_runtime_index(spaces: list[dict]) -> None:
global _INDEX, _INDEX_TS
texts = [str(space.get("search_text", "")) for space in spaces]
if not texts:
_INDEX = {
"signature": "",
"vectorizer": None,
"matrix": None,
"spaces": [],
}
_INDEX_TS = _now_ts()
return
vectorizer = TfidfVectorizer(ngram_range=(1, 2), lowercase=True, stop_words="english", max_features=12000, min_df=1)
matrix = vectorizer.fit_transform(texts)
signature = hashlib.sha1(json.dumps(texts, ensure_ascii=False).encode("utf-8")).hexdigest()
_INDEX = {
"signature": signature,
"vectorizer": vectorizer,
"matrix": matrix,
"spaces": spaces,
}
_INDEX_TS = _now_ts()
def _ensure_index(spaces: list[dict]) -> dict:
global _INDEX
texts = [str(space.get("search_text", "")) for space in spaces]
signature = hashlib.sha1(json.dumps(texts, ensure_ascii=False).encode("utf-8")).hexdigest()
if _INDEX and _INDEX.get("signature") == signature:
return _INDEX
_build_runtime_index(spaces)
return _INDEX or {}
def get_index_stats() -> dict:
load_spaces(force=False)
return dict(_STATS)
def get_category_distribution() -> dict[str, int]:
spaces = load_spaces(force=False)
counts: dict[str, int] = {}
for space in spaces:
category = str(space.get("category") or "Other")
counts[category] = counts.get(category, 0) + 1
return dict(sorted(counts.items(), key=lambda item: (-item[1], item[0].lower())))
def _extract_query_focus_terms(text: str) -> list[str]:
normalized = _normalize(text)
if not normalized:
return []
focus_map = [
(("kid", "child", "young", "family", "beginner", "friendly"), "kid-friendly"),
(("learn", "teach", "lesson", "study", "tutor", "education", "class"), "learning"),
(("story", "writing", "write", "novel", "poem", "creative", "journal"), "creative writing"),
(("image", "art", "design", "logo", "draw", "paint", "visual"), "image and design"),
(("voice", "audio", "speech", "transcrib", "music", "song", "sound"), "voice and speech"),
(("health", "wellness", "accessib", "dyslex", "therapy", "medical"), "health and accessibility"),
(("search", "rag", "knowledge", "retriev", "semantic", "embedding"), "search and knowledge"),
(("code", "developer", "api", "cli", "debug", "deploy", "github"), "developer tools"),
(("assistant", "agent", "productiv", "workflow", "task", "todo", "calendar"), "productivity and assistants"),
(("data", "research", "analysis", "analytics", "dataset", "csv", "sql"), "data and analysis"),
(("community", "social", "collaborat", "forum", "chat", "feedback", "team"), "community and collaboration"),
(("science", "math", "chemistry", "physics", "biology", "engineering"), "science and engineering"),
(("game", "quiz", "puzzle", "play", "interactive fiction", "trivia"), "games and puzzles"),
]
phrases: list[str] = []
for needles, phrase in focus_map:
if any(needle in normalized for needle in needles):
phrases.append(phrase)
return phrases[:4]
def _extract_evidence_snippet(source: str, query_terms: list[str]) -> str:
if not source:
return ""
source = re.sub(r"https?://\S+", "", source)
source = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", source)
sentences = [
re.sub(r"^\s*(?:[-*]|\d+[.)])\s+", "", part).strip()
for part in re.split(r"(?<=[.!?])\s+|\n+", source)
if part.strip()
]
lowered_terms = [term.lower() for term in query_terms if term]
generic = {
"about",
"overview",
"description",
"features",
"usage",
"quickstart",
"getting started",
"example",
"examples",
"model",
"dataset",
"inputs",
"outputs",
"intro",
}
noisy_markers = [
"field notes",
"blog post",
"demo video",
"watch demo",
"social post",
"github",
"achievement:",
"track:",
"badge:",
"sponsor:",
"llama.cpp",
"llama cpp",
"minicpm",
"nemotron",
"transformers",
"gradio",
"sdk:",
"api",
"pip install",
"uvicorn",
"requirements.txt",
"docker",
"modal",
"created by",
"social post",
"demo links",
"team (",
"hugging face usernames",
"collection:",
"judging notes",
]
action_terms = {
"game",
"games",
"play",
"match",
"matching",
"memory",
"cards",
"card",
"tutor",
"teach",
"lesson",
"learn",
"learning",
"write",
"writing",
"journal",
"story",
"creative",
"voice",
"audio",
"speech",
"search",
"knowledge",
"design",
"image",
"assistant",
"agent",
}
best_sentence = ""
best_score = float("-inf")
for sentence in sentences:
lowered = sentence.lower()
normalized = _normalize(sentence)
if any(marker in lowered for marker in noisy_markers):
continue
if normalized in generic:
continue
term_hits = sum(1 for term in lowered_terms if term and term in lowered)
action_hits = sum(1 for term in action_terms if term in normalized)
score = term_hits * 5 + min(action_hits, 3) * 2
if any(verb in normalized for verb in (" lets ", " helps ", " allows ", " turns ", " built around ", " starts ", " speak ", " paste ", " move ")):
score += 2
if len(normalized) >= 28:
score += 1
if len(normalized.split()) <= 4:
score -= 3
if normalized.startswith(("created by", "team", "social post", "demo links", "collection")):
score -= 8
if "build small hackathon" in normalized and action_hits == 0 and term_hits == 0:
score -= 4
if score > best_score:
best_sentence = sentence[:220].strip()
best_score = score
if best_sentence:
return best_sentence
for sentence in sentences:
lowered = _normalize(sentence)
if any(marker in lowered for marker in noisy_markers):
continue
if len(sentence) >= 24 and lowered not in generic:
return sentence[:220].strip()
return sentences[0][:220].strip() if sentences else ""
def _evidence_snippet_rank(snippet: str, query_terms: list[str]) -> int:
normalized = _normalize(snippet)
if not normalized:
return -999
lowered = snippet.lower()
lowered_terms = [term.lower() for term in query_terms if term]
noisy_markers = [
"created by",
"social post",
"demo links",
"team",
"collection",
"dataset",
"github",
"watch demo",
"field notes",
"design + build",
"llama.cpp",
"llama cpp",
"minicpm",
"nemotron",
"transformers",
"gradio",
"modal",
"docker",
"api",
]
action_terms = {
"game",
"games",
"play",
"arcade",
"match",
"matching",
"memory",
"card",
"cards",
"tutor",
"teach",
"lesson",
"learn",
"learning",
"write",
"writing",
"journal",
"story",
"voice",
"whiteboard",
"search",
"knowledge",
"design",
"image",
"assistant",
"agent",
"quiz",
}
term_hits = sum(1 for term in lowered_terms if term and term in lowered)
action_hits = sum(1 for term in action_terms if term in normalized)
score = term_hits * 5 + min(action_hits, 3) * 2
if len(normalized) >= 28:
score += 1
if len(normalized.split()) <= 4:
score -= 3
if any(marker in lowered for marker in noisy_markers):
score -= 8
return score
def _overlap_hits(query_terms: Iterable[str], source_text: str) -> list[str]:
source = _normalize(source_text)
hits: list[str] = []
for term in query_terms:
if term and term in source and term not in hits:
hits.append(term)
return hits
def _extract_code_evidence(code: str, limit: int = 4) -> list[str]:
if not code:
return []
lines = [line.rstrip() for line in code.splitlines() if line.strip()]
if not lines:
return []
score_patterns = [
"def ",
"class ",
"gr.",
"gradio",
"button",
"card",
"board",
"flip",
"match",
"pair",
"level",
"game",
"memory",
"quiz",
"chat",
"rerank",
"recommend",
"prompt",
]
selected: list[str] = []
for line in lines:
lowered = line.lower()
if any(pattern in lowered for pattern in score_patterns):
cleaned = re.sub(r"\s+", " ", line).strip()
if cleaned and cleaned not in selected:
selected.append(cleaned[:220])
if len(selected) >= limit:
break
if selected:
return selected[:limit]
return [re.sub(r"\s+", " ", line).strip()[:220] for line in lines[:limit] if line.strip()]
def _build_reason_brief(space: dict, user_query: str, liked_text: str = "") -> tuple[str, str]:
profile = _build_intent_profile(user_query, liked_text)
readme_summary = str(space.get("readme_summary", ""))
desc = str(space.get("desc", ""))
semantic_profile = space.get("semantic_profile") if isinstance(space.get("semantic_profile"), dict) else {}
semantic_text = _semantic_profile_text(semantic_profile) if semantic_profile else ""
readme_snippets = _query_aligned_evidence_snippets(space, user_query, liked_text, profile, limit=3)
if not readme_snippets and semantic_profile:
readme_snippets.extend(
[
str(semantic_profile.get("primary_activity", "")).strip(),
str(semantic_profile.get("audience", "")).strip(),
" ".join(semantic_profile.get("core_actions", []) or [])[:220],
" ".join(semantic_profile.get("value_points", []) or [])[:220],
]
)
readme_snippets = [snippet for snippet in readme_snippets if snippet]
if not readme_snippets and readme_summary:
readme_snippets.append(readme_summary[:220])
brief_parts = [
f"User query: {user_query or 'none'}",
f"Intent summary: {_compose_intent_summary(profile, user_query)}",
"You are writing a short review-style recommendation for a discovery card.",
f"Space title: {space.get('name', '')}",
f"One-line space summary: {desc or readme_summary or 'not provided'}",
f"Track: {space.get('track', '') or 'unknown'}",
f"Category: {space.get('category', 'Other')}",
f"Tags: {', '.join(list(space.get('tags', []) or [])[:8]) or 'none'}",
]
if semantic_profile:
brief_parts.append(f"Semantic profile: {semantic_text or 'not available'}")
if liked_text.strip():
brief_parts.append(f"User liked: {liked_text.strip()}")
matched_signals = list(space.get("matched_signals", []) or space.get("matched_tags", []) or [])
if matched_signals:
brief_parts.append(f"Helpful signals: {', '.join(matched_signals[:6])}")
if readme_snippets:
brief_parts.append("Evidence notes:")
brief_parts.extend(f"- {snippet}" for snippet in readme_snippets)
brief_parts.append("Write like a curator reviewing why this Space would delight or help the user.")
brief_parts.append("Focus on the feel and concrete behavior of the app, not on metadata or technical wording.")
return "\n".join(brief_parts), "\n".join(readme_snippets)
def _infer_space_mechanic(space: dict, evidence_text: str) -> str:
text = _normalize(
" ".join(
[
str(space.get("name", "")),
str(space.get("desc", "")),
str(space.get("readme_summary", "")),
evidence_text,
]
)
)
mechanic_rules = [
(("whiteboard", "tutor", "lesson"), "a whiteboard tutoring app"),
(("tutor", "lesson", "student"), "a tutor-style learning app"),
(("poker", "hold", "table"), "a poker game with live table talk"),
(("deckbuilder", "draft", "card"), "a deckbuilder with card-driven runs"),
(("flip", "match", "pair", "memory", "card"), "a card-matching memory game"),
(("quiz", "trivia", "question", "answer"), "a quiz-style game"),
(("read", "child", "feedback", "sentence"), "a child-facing reading coach"),
(("cyber", "security", "sandbox", "defensive"), "a defensive cybersecurity tool"),
(("write", "story", "blog", "journal"), "a writing and storytelling tool"),
(("story", "child", "bedtime", "illustrated"), "a children's story app"),
(("story", "novel", "narrative", "interactive fiction"), "an interactive storytelling app"),
(("image", "draw", "edit", "design", "canvas"), "an image or design tool"),
(("voice", "audio", "speech", "transcribe"), "a voice and audio app"),
(("search", "retrieve", "rag", "knowledge"), "a search and knowledge tool"),
(("chat", "assistant", "agent", "copilot"), "an assistant-style app"),
(("learn", "teach", "lesson", "student"), "a learning-focused app"),
(("game", "play", "level", "score"), "a game with interactive challenges"),
]
for needles, label in mechanic_rules:
if all(needle in text for needle in needles[:2]) or sum(1 for needle in needles if needle in text) >= 3:
return label
if "writing" in text or "journal" in text or "blog" in text:
return "a writing-focused app"
if "security" in text or "cyber" in text or "privacy" in text:
return "a security-focused app"
if "card" in text:
return "a card-based interactive app"
return "an app with mechanics that overlap with the query"
def build_recommendation_reason(space: dict, user_query: str, liked_text: str = "") -> str:
if not (user_query or "").strip():
return ""
space = hydrate_space_evidence(space, force=False)
brief, evidence = _build_reason_brief(space, user_query, liked_text)
llm_service = get_llm_service()
reason = llm_service.rewrite_recommendation_reason(
query=user_query,
repo_id=str(space.get("id", "")),
title=str(space.get("name", "")),
summary=str(space.get("desc", "")) or str(space.get("readme_summary", "")),
track=str(space.get("track", "")),
zone=str(space.get("category", "Other")),
tags=list(space.get("tags", []) or []),
evidence=brief if brief else evidence,
matched_signals=list(space.get("matched_signals", []) or space.get("matched_tags", []) or []),
liked_text=liked_text,
semantic_profile=dict(space.get("semantic_profile", {}) or {}),
readme_text=str(space.get("readme", "")),
)
if reason:
return reason
return ""
def _keyword_overlap_score(query_terms: list[str], text: str) -> tuple[int, list[str]]:
hits = _overlap_hits(query_terms, text)
return len(hits), hits
def recommend_spaces(
user_query: str,
liked_text: str = "",
category: str = "All",
top_k: int = 8,
) -> list[dict]:
spaces = filter_public_spaces(load_spaces(force=False))
if not spaces:
return []
top_k = max(1, min(int(top_k or 8), len(spaces)))
query = _normalize(user_query)
liked = _normalize(liked_text)
query_terms = _query_terms(query)
liked_terms = _query_terms(liked)
effective_query_terms = _expanded_query_terms(query_terms)
domains = _query_domains(query_terms)
intent_profile = _build_intent_profile(user_query, liked_text)
index = _ensure_index(spaces)
vectorizer = index.get("vectorizer")
matrix = index.get("matrix")
concept_query_text = " ".join(_query_alignment_terms(user_query, liked_text, intent_profile))
query_text = " ".join(part for part in [query, liked, concept_query_text] if part).strip()
retrieval_scores = np.zeros(len(spaces), dtype=float)
if vectorizer is not None and matrix is not None and query_text:
query_vec = vectorizer.transform([query_text])
retrieval_scores = cosine_similarity(matrix, query_vec).ravel()
if query_text:
candidate_indices = list(np.argsort(-retrieval_scores)[:40])
if intent_profile["required_concepts"]:
strict_matches = [
idx
for idx, space in enumerate(spaces)
if not _space_intent_alignment(space, intent_profile)["missing_required"]
]
soft_matches = [
idx
for idx, space in enumerate(spaces)
if _space_intent_alignment(space, intent_profile)["matched_required"]
]
candidate_indices = list(dict.fromkeys(strict_matches + soft_matches + candidate_indices))
if category != "All":
category_candidates = [idx for idx in candidate_indices if spaces[idx].get("category") == category]
if category_candidates:
candidate_indices = category_candidates
else:
candidate_indices = [idx for idx, space in enumerate(spaces) if space.get("category") == category]
else:
candidate_indices = [
idx for idx, space in enumerate(spaces) if category == "All" or space.get("category") == category
]
if not candidate_indices:
candidate_indices = list(range(len(spaces)))
candidate_indices = sorted(
candidate_indices,
key=lambda idx: (-int(spaces[idx].get("likes", 0) or 0), spaces[idx].get("name", "").lower()),
)
heuristic_rows: list[dict] = []
for idx in candidate_indices:
space = spaces[idx]
alignment = _space_intent_alignment(space, intent_profile)
required_categories = set(intent_profile.get("required_categories", set()) or set())
if category != "All" and space.get("category") != category:
category_score = 0.0
else:
category_score = 1.0 if category != "All" else 0.0
title_score, title_hits = _keyword_overlap_score(effective_query_terms, f"{space.get('name', '')} {space.get('raw_name', '')}")
desc_score, desc_hits = _keyword_overlap_score(effective_query_terms, space.get("desc", ""))
tag_score, tag_hits = _keyword_overlap_score(effective_query_terms, " ".join(space.get("tags", [])))
summary_score, summary_hits = _keyword_overlap_score(effective_query_terms, space.get("readme_summary", ""))
readme_score, readme_hits = _keyword_overlap_score(effective_query_terms, space.get("readme", ""))
liked_score, liked_hits = _keyword_overlap_score(liked_terms, space.get("search_text", ""))
domain_score, domain_signals = _domain_signal_score(space, domains)
tfidf_score = float(retrieval_scores[idx]) if idx < len(retrieval_scores) else 0.0
likes = int(space.get("likes", 0) or 0)
likes_bonus = min(np.log1p(likes) / 20.0, 0.18)
intent_score = len(alignment["matched_required"]) * 4.0 + len(alignment["matched_preferred"]) * 1.1
intent_penalty = len(alignment["missing_required"]) * 5.5
score = (
title_score * 3.0
+ desc_score * 1.25
+ tag_score * 2.25
+ summary_score * 2.0
+ readme_score * 1.5
+ liked_score * 0.9
+ domain_score
+ category_score * 2.5
+ tfidf_score * 4.0
+ intent_score
- intent_penalty
+ likes_bonus
)
if query_text:
evidence_count = title_score + desc_score + tag_score + summary_score + readme_score + liked_score
if tfidf_score < 0.015 and evidence_count <= 0:
continue
if "security" in domains and domain_score <= 0 and summary_score + readme_score + tag_score <= 0:
continue
if intent_profile["required_concepts"] and not alignment["matched_required"]:
continue
if required_categories and space.get("category") not in required_categories:
continue
if len(intent_profile["required_concepts"]) >= 2 and alignment["missing_required"]:
continue
matched_signals: list[str] = []
if alignment["matched_required"]:
matched_signals.append(f"intent: {', '.join(_ordered_concepts(alignment['matched_required'])[:3])}")
if title_hits:
matched_signals.append(f"query: {', '.join(title_hits[:2])}")
elif desc_hits:
matched_signals.append(f"query: {', '.join(desc_hits[:2])}")
if category != "All" and space.get("category") == category:
matched_signals.append(f"category: {category}")
if tag_hits:
matched_signals.append(f"tags: {', '.join(tag_hits[:3])}")
if summary_hits:
matched_signals.append(f"readme: {', '.join(summary_hits[:2])}")
if readme_hits and not summary_hits:
matched_signals.append(f"readme body: {', '.join(readme_hits[:2])}")
if liked_hits:
matched_signals.append("liked text matched")
if domain_signals:
matched_signals.append(f"domain: {', '.join(domain_signals[:3])}")
heuristic_rows.append(
{
"id": space.get("id", ""),
"name": space.get("name", ""),
"raw_name": space.get("raw_name", ""),
"author": space.get("author", ""),
"url": space.get("url", ""),
"category": space.get("category", "Other"),
"track": space.get("track", ""),
"description": space.get("desc", ""),
"tags": space.get("tags", []),
"likes": likes,
"matched_signals": matched_signals,
"reason": "",
"_heuristic_score": score,
}
)
if not heuristic_rows:
return []
heuristic_rows.sort(key=lambda item: (-item["_heuristic_score"], -item["likes"], item["name"].lower()))
llm_service = get_llm_service()
llm_map: dict[str, dict] = {}
llm_used = False
space_lookup = {str(space.get("id", "")): space for space in spaces if str(space.get("id", ""))}
hydrated_lookup: dict[str, dict] = {}
if query_text and heuristic_rows:
hydrated_lookup = {
repo_id: hydrate_space_evidence(space_lookup[repo_id], force=False)
for repo_id in {row["id"] for row in heuristic_rows[:30] if row["id"] in space_lookup}
}
try:
profile = llm_service.generate_query_profile(query_text, spaces)
except Exception as exc:
logger.warning("LLM query profile failed: %s", exc)
profile = {"summary": query_text, "confidence": 0.0, "primary_track": None, "keywords": [], "must_have": [], "avoid": []}
try:
prompt_candidates = [
{
"repo_id": row["id"],
"title": row["name"],
"summary": row["description"],
"track": row["track"],
"category": row["category"],
"tags": row["tags"],
"likes": row["likes"],
"readme_excerpt": (
"\n".join(
_query_aligned_evidence_snippets(
hydrated_lookup.get(row["id"], {}),
user_query,
liked_text,
intent_profile,
limit=3,
)
)[:650]
if row["id"] in hydrated_lookup
else ""
),
"semantic_profile": hydrated_lookup.get(row["id"], {}).get("semantic_profile", {}),
"matched_signals": row["matched_signals"],
}
for row in heuristic_rows[:12]
]
reranked = llm_service.rerank_candidates(query_text, prompt_candidates, profile=profile, limit=min(top_k * 2, 8))
llm_used = bool(reranked)
for item in reranked:
repo_id = str(item.get("repo_id", "")).strip()
if repo_id:
llm_map[repo_id] = item
if query_text and reranked == [] and getattr(llm_service, "available", lambda: False)():
return []
except Exception as exc:
logger.warning("LLM rerank failed: %s", exc)
llm_map = {}
llm_used = False
else:
profile = {"summary": query_text, "confidence": 0.0, "primary_track": None, "keywords": [], "must_have": [], "avoid": []}
max_heuristic = max((row["_heuristic_score"] for row in heuristic_rows), default=0.0)
results: list[dict] = []
for row in heuristic_rows:
llm_item = llm_map.get(row["id"], {})
heuristic_norm = (row["_heuristic_score"] / max_heuristic) if max_heuristic > 0 else 0.0
llm_norm = float(llm_item.get("rank_score", 0) or 0) / 100.0 if llm_item else 0.0
combined = heuristic_norm if not llm_used else (0.65 * heuristic_norm + 0.35 * llm_norm)
if query_text and combined < 0.22:
continue
result_reason = str(llm_item.get("reason", "")).strip() if llm_item else ""
results.append(
{
"id": row["id"],
"name": row["name"],
"url": row["url"],
"category": row["category"],
"description": row["description"],
"reason": result_reason,
"matched_signals": row["matched_signals"],
"tags": row["tags"],
"likes": row["likes"],
"track": row.get("track", ""),
"retrieval_score": float(row.get("_heuristic_score", 0.0)),
"_score": combined,
}
)
results.sort(key=lambda item: (-item["_score"], -item["likes"], item["name"].lower()))
final_results = results[:top_k]
if query_text:
for item in final_results:
source_space = hydrated_lookup.get(item["id"])
if source_space is None:
source_space = hydrate_space_evidence(space_lookup.get(item["id"], {}), force=True)
if source_space:
reason_space = dict(source_space)
reason_space["matched_signals"] = list(item.get("matched_signals", []) or [])
item["reason"] = build_recommendation_reason(reason_space, user_query=user_query, liked_text=liked_text)
else:
item["reason"] = ""
else:
for item in final_results:
item["reason"] = ""
return [
{
"id": item["id"],
"name": item["name"],
"url": item["url"],
"category": item["category"],
"description": item["description"],
"reason": item["reason"],
"matched_signals": item["matched_signals"],
"tags": item["tags"],
"likes": item["likes"],
"track": item.get("track", ""),
}
for item in final_results
]
def browse_spaces(category: str, limit: int = 12) -> list[dict]:
return recommend_spaces("", "", category=category, top_k=limit)
def refresh_spaces() -> list[dict]:
if _refresh_live_catalog_stats(force=True):
return [dict(item) for item in (_CACHE or [])]
return load_spaces(force=False)