| |
| """Build a static cross-dataset source-document and scaffold viewer.""" |
|
|
| from __future__ import annotations |
|
|
| import argparse |
| import csv |
| import hashlib |
| import json |
| import shutil |
| from pathlib import Path |
| from typing import Any, Iterable |
|
|
|
|
| ROOT = Path(__file__).resolve().parents[1] |
| INFO = Path("/home/azureuser/projects/information-scaffolds") |
| SESSION_INPUTS = Path( |
| "/home/azureuser/.copilot/session-state/6a2da107-0228-4316-ad25-f248cf796604/" |
| "files/structure-generation-viewer-inputs" |
| ) |
| DATA_ROOT = Path("/mnt/ramdisk/blobstore/timchen0618/data") |
|
|
| SHAPE_ALIASES = { |
| "tabular_records": "tables", |
| "tables": "tables", |
| "relation_graphs_and_mappings": "knowledge_graphs", |
| "knowledge_graphs": "knowledge_graphs", |
| "claim_and_theme_summaries": "claims", |
| "claims": "claims", |
| "chronology_and_timeline_indexes": "timelines", |
| "timelines": "timelines", |
| "qa_shortcuts_and_templates": "legacy_qa_shortcuts", |
| } |
|
|
| SHAPE_LABELS = { |
| "tables": "Tables", |
| "knowledge_graphs": "Knowledge graphs", |
| "claims": "Claims", |
| "timelines": "Timelines", |
| "legacy_qa_shortcuts": "Legacy QA shortcuts", |
| } |
|
|
| DATASETS = { |
| "monaco_dev": { |
| "label": "MoNaCo-dev", |
| "source": { |
| "kind": "raw_dir", |
| "path": INFO / "outputs/rawtext_corpus_monacodev/monaco_dev/scaffolds", |
| }, |
| "runs": [ |
| { |
| "id": "legacy_full", |
| "label": "Legacy full extraction - 50-doc subset", |
| "job": "sweet_screw_frnspf41x5", |
| "schema": "legacy five-shape", |
| "root": INFO |
| / "outputs/e2e_dev_scaffolds/monaco_dev/named-outputs/scaffolds_dir", |
| }, |
| { |
| "id": "current", |
| "label": "Current schema-v3 extraction", |
| "job": "quirky_cabbage_zdpp543s3z", |
| "schema": "3", |
| "root": SESSION_INPUTS / "monaco_dev/current/scaffolds_dir", |
| "quality": SESSION_INPUTS / "monaco_dev/current/schema_quality_report/report", |
| }, |
| ], |
| }, |
| "wiki_opentable_dev": { |
| "label": "Open-WikiTable-dev", |
| "source": { |
| "kind": "raw_dir", |
| "path": INFO |
| / "outputs/rawtext_corpus_wikiotdev/wiki_opentable_dev/scaffolds", |
| }, |
| "runs": [ |
| { |
| "id": "legacy_full", |
| "label": "Legacy full extraction - 50-doc subset", |
| "job": "cyan_muscle_ylws4v86bz", |
| "schema": "legacy five-shape", |
| "root": INFO |
| / "outputs/e2e_dev_scaffolds/wiki_opentable_dev/named-outputs/scaffolds_dir", |
| }, |
| { |
| "id": "current", |
| "label": "Current schema-v3 extraction", |
| "job": "purple_eye_8f31ymzz71", |
| "schema": "3", |
| "root": SESSION_INPUTS / "wiki_opentable_dev/current/scaffolds_dir", |
| "quality": SESSION_INPUTS |
| / "wiki_opentable_dev/current/schema_quality_report/report", |
| }, |
| ], |
| }, |
| "phantom_wiki": { |
| "label": "PhantomWiki", |
| "source": { |
| "kind": "corpus", |
| "path": DATA_ROOT / "eval/phantom_wiki/unified/corpus.unified.jsonl", |
| }, |
| "runs": [ |
| { |
| "id": "legacy_full", |
| "label": "Legacy full extraction - 50-doc subset", |
| "job": "teal_jicama_qxvpyf3jsd", |
| "schema": "legacy five-shape", |
| "root": INFO |
| / "outputs/e2e_runs/new-datasets-full-20260711/phantom_wiki/" |
| "named-outputs/scaffolds_dir", |
| }, |
| { |
| "id": "current", |
| "label": "Current schema-v3 extraction", |
| "job": "mango_avocado_7qnv0vlf46", |
| "schema": "3", |
| "root": SESSION_INPUTS / "phantom_wiki/current/scaffolds_dir", |
| "quality": SESSION_INPUTS |
| / "phantom_wiki/current/schema_quality_report/report", |
| }, |
| ], |
| }, |
| } |
|
|
|
|
| def read_json(path: Path) -> Any: |
| return json.loads(path.read_text(encoding="utf-8")) |
|
|
|
|
| def read_jsonl(path: Path) -> Iterable[dict[str, Any]]: |
| with path.open(encoding="utf-8") as handle: |
| for line_number, line in enumerate(handle, 1): |
| if not line.strip(): |
| continue |
| value = json.loads(line) |
| if not isinstance(value, dict): |
| raise ValueError(f"{path}:{line_number}: expected an object") |
| yield value |
|
|
|
|
| def parse_doc_id(text: str, fallback: str) -> str: |
| for raw in text.splitlines(): |
| line = raw.strip() |
| if not line: |
| continue |
| if line.startswith("#") and line.lstrip("#").strip().lower().startswith("id:"): |
| value = line.lstrip("#").strip().split(":", 1)[1].strip() |
| return value or fallback |
| return fallback |
| return fallback |
|
|
|
|
| def load_selected_sources(source: dict[str, Any]) -> list[dict[str, str]]: |
| path = Path(source["path"]) |
| if source["kind"] == "raw_dir": |
| selected: list[Path] = [] |
| for candidate in path.rglob("*.txt"): |
| if candidate.is_file(): |
| selected.append(candidate) |
| if len(selected) == 50: |
| break |
| selected.sort() |
| rows = [] |
| for candidate in selected: |
| text = candidate.read_text(encoding="utf-8") |
| rows.append( |
| { |
| "doc_id": parse_doc_id(text, candidate.stem), |
| "contents": text, |
| "source_path": str(candidate.relative_to(path)), |
| } |
| ) |
| return rows |
|
|
| rows = [] |
| for row in read_jsonl(path): |
| rows.append( |
| { |
| "doc_id": str(row["id"]), |
| "contents": str(row.get("contents", "")), |
| "source_path": str(row["id"]), |
| } |
| ) |
| if len(rows) == 50: |
| break |
| return rows |
|
|
|
|
| def artifact_format(filename: str) -> str: |
| if filename.endswith(".edges.jsonl"): |
| return "jsonl" |
| if filename.endswith(".timeline.json"): |
| return "json" |
| suffix = Path(filename).suffix.lower().lstrip(".") |
| return suffix if suffix in {"csv", "json", "jsonl", "md"} else "text" |
|
|
|
|
| def load_quality(path: Path | None) -> dict[str, Any] | None: |
| if path is None or not path.exists(): |
| return None |
| return read_json(path) |
|
|
|
|
| def load_scaffold_run(config: dict[str, Any]) -> dict[str, Any]: |
| if config.get("status") == "pending": |
| return { |
| "id": config["id"], |
| "label": config["label"], |
| "job": config["job"], |
| "schema": config["schema"], |
| "status": "pending", |
| "documents": {}, |
| "quality": None, |
| } |
|
|
| root = Path(config["root"]) |
| top = read_json(root / "_index.json") |
| documents: dict[str, list[dict[str, Any]]] = {} |
| shape_summaries = [] |
| for shape in top.get("shapes", []): |
| raw_shape = str(shape["shape_id"]) |
| canonical_shape = SHAPE_ALIASES.get(raw_shape, raw_shape) |
| folder = str(shape.get("folder") or raw_shape).rstrip("/") |
| shape_index = read_json(root / folder / "_index.json") |
| shape_summaries.append( |
| { |
| "id": canonical_shape, |
| "raw_id": raw_shape, |
| "label": SHAPE_LABELS.get(canonical_shape, shape.get("display_name", raw_shape)), |
| "description": shape_index.get("description", shape.get("definition", "")), |
| "n_files": len(shape_index.get("entries", [])), |
| } |
| ) |
| for entry in shape_index.get("entries", []): |
| doc_id = str(entry["doc_id"]) |
| filename = str(entry["file"]) |
| artifact_path = root / folder / filename |
| documents.setdefault(doc_id, []).append( |
| { |
| "shape": canonical_shape, |
| "shape_label": SHAPE_LABELS.get(canonical_shape, canonical_shape), |
| "raw_shape": raw_shape, |
| "filename": filename, |
| "format": artifact_format(filename), |
| "unit_name": entry.get("unit_name"), |
| "unit_description": entry.get("unit_description"), |
| "content": artifact_path.read_text(encoding="utf-8"), |
| } |
| ) |
|
|
| return { |
| "id": config["id"], |
| "label": config["label"], |
| "job": config["job"], |
| "schema": config["schema"], |
| "status": "ready", |
| "documents": documents, |
| "quality": load_quality(config.get("quality")), |
| "summary": { |
| "n_docs_seen": top.get("n_docs_seen"), |
| "n_files_written": top.get("n_files_written"), |
| "shapes_hash": top.get("shapes_hash"), |
| "shape_summaries": shape_summaries, |
| "validation_metrics": top.get("validation_metrics"), |
| }, |
| } |
|
|
|
|
| def record_name(doc_id: str) -> str: |
| return hashlib.sha1(doc_id.encode()).hexdigest() + ".json" |
|
|
|
|
| def build_dataset(dataset_id: str, config: dict[str, Any], data_dir: Path) -> dict[str, Any]: |
| sources = load_selected_sources(config["source"]) |
| if len(sources) != 50: |
| raise ValueError(f"{dataset_id}: expected 50 source documents, got {len(sources)}") |
| source_ids = [row["doc_id"] for row in sources] |
| if len(set(source_ids)) != 50: |
| raise ValueError(f"{dataset_id}: selected source IDs are not unique") |
|
|
| runs = [load_scaffold_run(run) for run in config["runs"]] |
| records_dir = data_dir / "records" / dataset_id |
| records_dir.mkdir(parents=True, exist_ok=True) |
| index_rows = [] |
| for position, source in enumerate(sources, 1): |
| doc_id = source["doc_id"] |
| run_records = {} |
| for run in runs: |
| artifacts = run["documents"].get(doc_id, []) |
| run_records[run["id"]] = { |
| "status": run["status"], |
| "artifacts": artifacts, |
| "n_artifacts": len(artifacts), |
| } |
| digest = hashlib.sha256(source["contents"].encode()).hexdigest() |
| record = { |
| "dataset": dataset_id, |
| "dataset_label": config["label"], |
| "position": position, |
| "doc_id": doc_id, |
| "source": { |
| "contents": source["contents"], |
| "source_path": source["source_path"], |
| "sha256": digest, |
| }, |
| "runs": run_records, |
| } |
| filename = record_name(doc_id) |
| (records_dir / filename).write_text( |
| json.dumps(record, ensure_ascii=False), encoding="utf-8" |
| ) |
| index_rows.append( |
| { |
| "position": position, |
| "doc_id": doc_id, |
| "record": f"data/records/{dataset_id}/{filename}", |
| "source_preview": source["contents"].replace("\n", " ")[:180], |
| "source_sha256": digest, |
| "run_counts": { |
| run_id: value["n_artifacts"] for run_id, value in run_records.items() |
| }, |
| } |
| ) |
|
|
| return { |
| "id": dataset_id, |
| "label": config["label"], |
| "n_docs": len(sources), |
| "runs": [ |
| { |
| key: run.get(key) |
| for key in ("id", "label", "job", "schema", "status", "quality", "summary") |
| } |
| for run in runs |
| ], |
| "records": index_rows, |
| } |
|
|
|
|
| def validate_bundle(index: dict[str, Any], data_dir: Path) -> None: |
| if len(index["datasets"]) != 3: |
| raise ValueError("expected three datasets") |
| for dataset in index["datasets"]: |
| if dataset["n_docs"] != 50 or len(dataset["records"]) != 50: |
| raise ValueError(f"{dataset['id']}: expected 50 records") |
| if len({row["doc_id"] for row in dataset["records"]}) != 50: |
| raise ValueError(f"{dataset['id']}: duplicate document IDs") |
| for row in dataset["records"]: |
| path = ROOT / row["record"] |
| if not path.exists(): |
| raise ValueError(f"missing record shard: {path}") |
| phantom = next(row for row in index["datasets"] if row["id"] == "phantom_wiki") |
| current = next(run for run in phantom["runs"] if run["id"] == "current") |
| if current["status"] != "ready": |
| raise ValueError("PhantomWiki current run must be ready") |
|
|
|
|
| def main() -> int: |
| parser = argparse.ArgumentParser(description=__doc__) |
| parser.add_argument("--out", type=Path, default=ROOT / "data") |
| args = parser.parse_args() |
|
|
| if args.out.exists(): |
| shutil.rmtree(args.out) |
| args.out.mkdir(parents=True) |
| index = { |
| "title": "Structure Generation Prompt Viewer", |
| "datasets": [ |
| build_dataset(dataset_id, config, args.out) |
| for dataset_id, config in DATASETS.items() |
| ], |
| } |
| (args.out / "index.json").write_text( |
| json.dumps(index, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" |
| ) |
| validate_bundle(index, args.out) |
| print( |
| "Built 150 document records: " |
| + ", ".join(f"{row['label']}={row['n_docs']}" for row in index["datasets"]) |
| ) |
| return 0 |
|
|
|
|
| if __name__ == "__main__": |
| raise SystemExit(main()) |
|
|