#!/usr/bin/env python3 """Audit locally available formula/description evidence and emit an AI review queue. The queue contains source text and identity candidates only. It deliberately does not generate perceptual labels, assessor outcomes, or chemical identities. Those require a separately reviewed, provenance-preserving decision. """ from __future__ import annotations import argparse import json from pathlib import Path from typing import Any ROOT = Path(__file__).resolve().parents[1] DATA = ROOT / "data" def load_jsonl(path: Path) -> list[dict[str, Any]]: if not path.exists(): return [] return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()] def formula_description(row: dict[str, Any]) -> str: source = row.get("source") if isinstance(row.get("source"), dict) else {} return str( row.get("description") or row.get("profile_text") or row.get("overall") or source.get("description") or "" ).strip() def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--output", type=Path, default=ROOT / "artifacts/source_population_audit.json") parser.add_argument("--ai-review-queue", type=Path, default=ROOT / "artifacts/source_population_ai_review.jsonl") args = parser.parse_args() sources = [ DATA / "wisemoor_free_formulas.sanitized.jsonl", DATA / "tgsc_demo_formulas_v2.jsonl", DATA / "literature_formulas_poucher_with_profiles.jsonl", DATA / "appell_formulas_enriched.jsonl", ] inventory: list[dict[str, Any]] = [] review: list[dict[str, Any]] = [] for path in sources: rows = load_jsonl(path) described = [(index, row) for index, row in enumerate(rows) if formula_description(row)] inventory.append({ "source": str(path.relative_to(ROOT)), "formula_rows": len(rows), "rows_with_source_description": len(described), }) for index, row in described: source = row.get("source") if isinstance(row.get("source"), dict) else {} review.append({ "review_id": f"{path.stem}:{index}", "task": "extract_formula_level_attributes_with_verbatim_source_spans", "source_file": str(path.relative_to(ROOT)), "source_row_index": index, "formula_id": row.get("formula_id") or row.get("id") or row.get("product_id"), "formula_name": row.get("name") or source.get("title"), "source_text": formula_description(row), "allowed_output": ["attribute", "source_span", "confidence", "abstain_reason"], "forbidden_output": ["assessor_label", "invented_identity", "invented_metric"], "status": "pending_review", }) tgsc_cache_path = DATA / "tgsc_odor_cache.json" tgsc_cache = json.loads(tgsc_cache_path.read_text(encoding="utf-8")) if tgsc_cache_path.exists() else {} profiles = load_jsonl(DATA / "material_profiles_v11_5.jsonl") report = { "schema_version": 1, "formula_sources": inventory, "formula_rows": sum(item["formula_rows"] for item in inventory), "formula_rows_with_source_description": sum(item["rows_with_source_description"] for item in inventory), "material_profiles": len(profiles), "material_profiles_with_odor_description": sum(bool(row.get("odor_descriptions")) for row in profiles), "tgsc_cached_material_pages": len(tgsc_cache), "ai_review_rows": len(review), "policy": "AI may extract claims only with verbatim source spans; outputs remain proposals until reviewed and are never assessor truth.", } args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") args.ai_review_queue.parent.mkdir(parents=True, exist_ok=True) with args.ai_review_queue.open("w", encoding="utf-8") as handle: for row in review: handle.write(json.dumps(row, ensure_ascii=False, sort_keys=True) + "\n") print(json.dumps(report, indent=2, sort_keys=True)) if __name__ == "__main__": main()