Spaces:
Sleeping
Sleeping
| """Quarterly topic lifecycle (Gartner hype-cycle) snapshots for the CS arXiv set. | |
| 参考 Elfsong/hf_paper_lifecycle 的相位分类算法,但: | |
| - 时间单位用【季度 Q】(每 3 个月一桶),而非月/双月; | |
| - 数据源是本地 arxiv_cs_2022_2026.topics.jsonl 的 openalex_topics; | |
| - slope 用 numpy.polyfit(免 scipy),推送用 huggingface_hub 直接传文件(免 datasets)。 | |
| 对每个季度 Q 产出一份"累积到 Q 为止"的快照(slider 拖到某季度即看该快照), | |
| 所有快照打包进 lifecycle_quarterly.json,供 app.py 可视化,并可 --push 到 HF。 | |
| Usage: | |
| uv run python src/lifecycle_quarterly.py # 计算并写本地 json | |
| uv run python src/lifecycle_quarterly.py --push # 同时上传 HF | |
| uv run python src/lifecycle_quarterly.py --repo Elfsong/xxx --push | |
| """ | |
| import argparse | |
| import json | |
| import os | |
| from collections import Counter, defaultdict | |
| from pathlib import Path | |
| import numpy as np | |
| from dotenv import load_dotenv | |
| ROOT = Path(__file__).resolve().parent.parent | |
| load_dotenv(ROOT / ".env") | |
| SRC = ROOT / "arxiv_cs_2022_2026.topics.jsonl" | |
| OUT = ROOT / "lifecycle_quarterly.json" | |
| TOP_COUNT = 160 # 每快照按累计论文数取的"成熟"topic 数 | |
| TOP_EMERGING = 40 # 额外按"新兴度"补的 topic 数(让 Innovation 非空) | |
| PHASE_ORDER = [ | |
| "Innovation Trigger", | |
| "Peak of Inflated Expectations", | |
| "Trough of Disillusionment", | |
| "Slope of Enlightenment", | |
| "Plateau of Productivity", | |
| ] | |
| def quarter_of(rec: dict): | |
| ds = rec.get("submitted_date") | |
| if ds and len(ds) >= 7 and ds[:4].isdigit(): | |
| y, m = int(ds[:4]), int(ds[5:7]) | |
| else: # 退化:从 arXiv id 的 YYMM 取 | |
| pid = rec.get("id", "") | |
| if len(pid) >= 4 and pid[:4].isdigit(): | |
| y, m = 2000 + int(pid[:2]), int(pid[2:4]) | |
| else: | |
| return None | |
| if not (1 <= m <= 12): | |
| return None | |
| return f"{y}-Q{(m - 1) // 3 + 1}" | |
| def slope(y): | |
| if len(y) < 3: | |
| return 0.0 | |
| x = np.arange(len(y), dtype=float) | |
| return float(np.polyfit(x, y, 1)[0]) | |
| def classify(dr, sl, qa, qsp, tc, rf): | |
| """相位判定。阈值由参考脚本的"月"折算为"季度": | |
| 时间阈值 ÷3(8月≈3季度,6月≈2季度…),slope 阈值 ×3(季度步长约为月的 3 倍)。""" | |
| if qa <= 3 or (rf > 0.60 and tc < 200): | |
| return "Innovation Trigger" | |
| if (dr > 0.70 and qsp <= 2) or (sl > 0.003 and dr > 0.65): | |
| return "Peak of Inflated Expectations" | |
| if dr < 0.65: | |
| return "Slope of Enlightenment" if sl > 0.001 else "Trough of Disillusionment" | |
| if sl < -0.003 and dr < 0.75: | |
| return "Trough of Disillusionment" | |
| if dr < 0.85 and sl > 0.0015 and qsp > 1: | |
| return "Slope of Enlightenment" | |
| return "Plateau of Productivity" | |
| def lifecycle_for(quarters, tbq): | |
| """给定有序季度列表 + {quarter: Counter(topic->count)},算各 topic 指标。""" | |
| n = len(quarters) | |
| if n < 2: | |
| return [] | |
| total_by_q = {q: sum(tbq[q].values()) for q in quarters} | |
| all_topics = Counter() | |
| for q in quarters: | |
| all_topics.update(tbq[q]) | |
| min_papers = max(3, n) | |
| out = [] | |
| for topic, tot in all_topics.items(): | |
| if tot < min_papers: | |
| continue | |
| props = np.array([ | |
| (tbq[q].get(topic, 0) / total_by_q[q]) if total_by_q[q] else 0.0 | |
| for q in quarters | |
| ]) | |
| counts = np.array([tbq[q].get(topic, 0) for q in quarters]) | |
| nz = np.where(props > 0)[0] | |
| if len(nz) < 2: | |
| continue | |
| first_idx = int(nz[0]) | |
| peak_idx = int(np.argmax(props)) | |
| peak_val = float(props[peak_idx]) | |
| current_avg = float(np.mean(props[-min(3, n):])) | |
| recent = props[-min(4, n):] # slope 窗口:近 4 季度(≈1 年) | |
| sl = slope(recent) | |
| dr = current_avg / peak_val if peak_val > 0 else 0.0 | |
| qsp = n - 1 - peak_idx | |
| qa = n - first_idx | |
| rw = min(4, len(counts)) # recent_fraction 窗口:近 4 季度 | |
| rf = float(counts[-rw:].sum() / max(counts.sum(), 1)) | |
| phase = classify(dr, sl, qa, qsp, int(counts.sum()), rf) | |
| out.append({ | |
| "topic": topic, | |
| "phase": phase, | |
| "total_count": int(tot), | |
| "peak_val": round(peak_val, 6), | |
| "peak_quarter": quarters[peak_idx], | |
| "current_avg": round(current_avg, 6), | |
| "slope": round(sl, 8), | |
| "decline_ratio": round(dr, 4), | |
| "recent_fraction": round(rf, 4), | |
| "quarters_active": qa, | |
| "quarters_since_peak": qsp, | |
| "emerging": False, | |
| }) | |
| # 选择:top-N 成熟(按累计计数) ∪ top-M 新兴(Innovation 或近期占比高) | |
| out.sort(key=lambda r: r["total_count"], reverse=True) | |
| mature = out[:TOP_COUNT] | |
| chosen = {t["topic"] for t in mature} | |
| emerging_pool = [ | |
| t for t in out | |
| if t["topic"] not in chosen | |
| and (t["phase"] == "Innovation Trigger" or t["recent_fraction"] > 0.5) | |
| ] | |
| emerging_pool.sort(key=lambda r: r["total_count"], reverse=True) | |
| emerging = emerging_pool[:TOP_EMERGING] | |
| for t in emerging: | |
| t["emerging"] = True | |
| return mature + emerging | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--push", action="store_true") | |
| ap.add_argument("--repo", default=os.getenv("HF_LIFECYCLE_REPO", | |
| "Elfsong/arxiv_cs_lifecycle")) | |
| args = ap.parse_args() | |
| # 1) 一次性按季度聚合 topics | |
| tbq = defaultdict(Counter) | |
| n_by_q = Counter() | |
| n_total = n_topic = 0 | |
| with open(SRC) as f: | |
| for line in f: | |
| if not line.strip(): | |
| continue | |
| rec = json.loads(line) | |
| n_total += 1 | |
| q = quarter_of(rec) | |
| if not q: | |
| continue | |
| topics = rec.get("openalex_topics") or [] | |
| if topics: | |
| n_topic += 1 | |
| tbq[q].update(topics) | |
| n_by_q[q] += 1 | |
| quarters = sorted(tbq.keys()) | |
| print(f"papers={n_total} with_topics={n_topic} quarters={quarters}") | |
| # 2) 逐季度做累积快照 | |
| snapshots = {} | |
| for i, q in enumerate(quarters): | |
| sub_q = quarters[: i + 1] | |
| topics = lifecycle_for(sub_q, tbq) | |
| phase_counts = Counter(t["phase"] for t in topics) | |
| snapshots[q] = { | |
| "n_papers": sum(n_by_q[x] for x in sub_q), | |
| "quarters": sub_q, | |
| "phase_counts": {p: phase_counts.get(p, 0) for p in PHASE_ORDER}, | |
| "topics": topics, | |
| } | |
| print(f" {q}: {snapshots[q]['n_papers']} papers, {len(topics)} topics, " | |
| + ", ".join(f"{p.split()[0]}:{phase_counts.get(p,0)}" for p in PHASE_ORDER)) | |
| doc = { | |
| "source": SRC.name, | |
| "quarters": quarters, | |
| "phase_order": PHASE_ORDER, | |
| "snapshots": snapshots, | |
| } | |
| OUT.write_text(json.dumps(doc, ensure_ascii=False)) | |
| print(f"written -> {OUT} ({OUT.stat().st_size/1e6:.1f} MB)") | |
| if args.push: | |
| push_to_hf(OUT, args.repo) | |
| def push_to_hf(path: Path, repo: str): | |
| from huggingface_hub import HfApi | |
| token = os.getenv("HF_TOKEN", "") | |
| if not token: | |
| raise RuntimeError("HF_TOKEN not set in .env") | |
| api = HfApi(token=token) | |
| api.create_repo(repo, repo_type="dataset", exist_ok=True) | |
| api.upload_file(path_or_fileobj=str(path), path_in_repo=path.name, | |
| repo_id=repo, repo_type="dataset") | |
| print(f"pushed {path.name} -> https://huggingface.co/datasets/{repo}") | |
| if __name__ == "__main__": | |
| main() | |