Spaces:
Sleeping
Sleeping
| from __future__ import annotations | |
| import csv | |
| import json | |
| import mimetypes | |
| import os | |
| import re | |
| import shutil | |
| import tempfile | |
| import time | |
| from datetime import datetime, timezone | |
| from pathlib import Path | |
| from typing import Any, Iterable | |
| from .canonical_json import build_deterministic_zip, canonical_json_bytes, write_canonical_json, write_text_lf | |
| from .claim_ladder import bound_claims | |
| from .dpio_reader import build_dpio_read | |
| from .hashing import sha256_bytes, sha256_file | |
| from .metric_analysis import METRIC_HEADERS, analyze_distribution, distribution_report_markdown, parse_metric_receipts | |
| from .sensitive_data import scan_file, scan_text | |
| from .temporal_ratchet import sort_events | |
| from .validation import gate_receipt, validate_capsule, validate_causal_arc | |
| MAX_FILE_BYTES = 100 * 1024 * 1024 | |
| MAX_TOTAL_BYTES = 500 * 1024 * 1024 | |
| ALLOWED_EXTENSIONS = { | |
| ".png", ".jpg", ".jpeg", ".webp", ".gif", | |
| ".mp4", ".mov", ".webm", | |
| ".csv", ".json", ".txt", ".md", ".pdf", ".zip", | |
| } | |
| DEFAULT_ACKNOWLEDGEMENTS = [ | |
| "I understand this prototype preserves evidence but does not prove suppression, targeting, motive, or intent.", | |
| "I control whether the generated packet is shared.", | |
| "I have reviewed uploads for sensitive or unnecessary third-party information.", | |
| ] | |
| CLAIMS_NOT_TO_MAKE = [ | |
| "This packet alone proves shadowbanning or deliberate suppression.", | |
| "This packet alone identifies an internal classifier, employee, executive, or authorization chain.", | |
| "A missing notification alone proves deliberate withholding.", | |
| "A label removal alone proves full downstream restoration.", | |
| "A single account establishes motive or intent.", | |
| ] | |
| def cleanup_old_workspaces(max_age_seconds: int = 6 * 60 * 60) -> None: | |
| temp_root = Path(tempfile.gettempdir()) | |
| now = time.time() | |
| for candidate in temp_root.glob("substrate_creator_packet_*"): | |
| try: | |
| if candidate.is_dir() and now - candidate.stat().st_mtime > max_age_seconds: | |
| shutil.rmtree(candidate, ignore_errors=True) | |
| except OSError: | |
| continue | |
| def utc_now_iso() -> str: | |
| return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z") | |
| def parse_lines(value: Any) -> list[str]: | |
| if value is None: | |
| return [] | |
| if isinstance(value, list): | |
| return [str(x).strip() for x in value if str(x).strip()] | |
| text = str(value) | |
| return [line.strip(" \t-•") for line in text.splitlines() if line.strip(" \t-•")] | |
| def safe_slug(value: str, fallback: str = "CREATOR") -> str: | |
| cleaned = re.sub(r"[^A-Za-z0-9_.-]+", "_", value.strip()).strip("._-") | |
| return cleaned or fallback | |
| def capsule_id_from_seed(platform: str, handle: str, created_at: str, topology: str) -> str: | |
| seed = { | |
| "platform": platform.strip(), | |
| "handle": handle.strip(), | |
| "created_at": created_at.strip(), | |
| "creator_exact_description": topology, | |
| } | |
| suffix = sha256_bytes(canonical_json_bytes(seed))[:12].upper() | |
| prefix = safe_slug(f"{platform}_{handle}", "CREATOR").upper() | |
| return f"CREATOR_{prefix}_{suffix}" | |
| def _cell(row: Any, index: int, default: Any = "") -> Any: | |
| if row is None: | |
| return default | |
| if isinstance(row, dict): | |
| return default | |
| if index >= len(row): | |
| return default | |
| value = row[index] | |
| if value is None: | |
| return default | |
| return value | |
| def rows_from_dataframe(data: Any) -> list[list[Any]]: | |
| if data is None: | |
| return [] | |
| if hasattr(data, "values"): | |
| return data.fillna("").values.tolist() | |
| if isinstance(data, dict) and "data" in data: | |
| return data["data"] or [] | |
| if isinstance(data, list): | |
| return data | |
| return [] | |
| def parse_events(data: Any) -> list[dict[str, Any]]: | |
| rows = rows_from_dataframe(data) | |
| events: list[dict[str, Any]] = [] | |
| for row_number, row in enumerate(rows, start=1): | |
| description = str(_cell(row, 4)).strip() | |
| if not description: | |
| continue | |
| raw_index = _cell(row, 0, row_number - 1) | |
| try: | |
| sequence_index = int(float(raw_index)) | |
| except (TypeError, ValueError): | |
| sequence_index = row_number - 1 | |
| event_id = str(_cell(row, 9)).strip() or f"EVENT_{sequence_index:04d}" | |
| evidence_basis = parse_lines(str(_cell(row, 6)).replace(";", "\n")) | |
| corrects = str(_cell(row, 7)).strip() or None | |
| predecessors = parse_lines(str(_cell(row, 10)).replace(";", "\n")) | |
| event = { | |
| "event_id": event_id, | |
| "sequence_index": sequence_index, | |
| "event_type": str(_cell(row, 1, "OTHER")).strip() or "OTHER", | |
| "event_time": str(_cell(row, 2)).strip(), | |
| "observed_time": str(_cell(row, 3)).strip(), | |
| "description": description, | |
| "evidence_basis": evidence_basis, | |
| "predecessor_event_ids": predecessors, | |
| "corrects_event_id": corrects, | |
| "state": str(_cell(row, 5, "CREATOR_REPORTED")).strip() or "CREATOR_REPORTED", | |
| "uncertainty": str(_cell(row, 8)).strip(), | |
| } | |
| # Empty optional strings are valid under the schema, but omit no fields | |
| # that are used in deterministic exports. | |
| events.append(event) | |
| return sort_events(events) | |
| def parse_claims(data: Any) -> list[dict[str, Any]]: | |
| rows = rows_from_dataframe(data) | |
| claims: list[dict[str, Any]] = [] | |
| for row_number, row in enumerate(rows, start=1): | |
| statement = str(_cell(row, 1)).strip() | |
| if not statement: | |
| continue | |
| claims.append({ | |
| "claim_id": str(_cell(row, 6)).strip() or f"CLAIM_{row_number:04d}", | |
| "claim_level": str(_cell(row, 0, "L1_DIRECT_OBSERVATION")).strip() or "L1_DIRECT_OBSERVATION", | |
| "statement": statement, | |
| "basis": parse_lines(str(_cell(row, 2)).replace(";", "\n")), | |
| "state": str(_cell(row, 3, "CREATOR_REPORTED")).strip() or "CREATOR_REPORTED", | |
| "uncertainty": str(_cell(row, 4)).strip(), | |
| "human_review_required": str(_cell(row, 5)).strip().lower() in {"true", "yes", "1", "required"}, | |
| }) | |
| return bound_claims(claims) | |
| def _normalize_upload_paths(uploaded_files: Any) -> list[Path]: | |
| if uploaded_files is None: | |
| return [] | |
| items = uploaded_files if isinstance(uploaded_files, list) else [uploaded_files] | |
| paths: list[Path] = [] | |
| for item in items: | |
| if item is None: | |
| continue | |
| if isinstance(item, str): | |
| path = Path(item) | |
| elif hasattr(item, "name"): | |
| path = Path(item.name) | |
| elif isinstance(item, dict) and item.get("path"): | |
| path = Path(item["path"]) | |
| else: | |
| raise ValueError(f"Unsupported upload object: {type(item)!r}") | |
| if not path.exists() or not path.is_file(): | |
| raise ValueError(f"Uploaded file is unavailable: {path}") | |
| paths.append(path) | |
| return paths | |
| def _asset_metadata_map(metadata_json: str) -> dict[str, dict[str, Any]]: | |
| if not str(metadata_json or "").strip(): | |
| return {} | |
| parsed = json.loads(metadata_json) | |
| if not isinstance(parsed, dict): | |
| raise ValueError("Per-file metadata JSON must be an object keyed by original filename.") | |
| return parsed | |
| def _copy_and_manifest_assets( | |
| uploads: list[Path], | |
| evidence_dir: Path, | |
| created_at: str, | |
| event_ids: set[str], | |
| common: dict[str, Any], | |
| metadata_map: dict[str, dict[str, Any]], | |
| ) -> tuple[list[dict[str, Any]], list[str], bool]: | |
| total = sum(p.stat().st_size for p in uploads) | |
| if total > MAX_TOTAL_BYTES: | |
| raise ValueError(f"Total uploads exceed {MAX_TOTAL_BYTES // (1024*1024)} MB.") | |
| manifests: list[dict[str, Any]] = [] | |
| warnings: list[str] = [] | |
| must_stop = False | |
| used_names: set[str] = set() | |
| evidence_dir.mkdir(parents=True, exist_ok=True) | |
| for index, source in enumerate(uploads, start=1): | |
| size = source.stat().st_size | |
| if size > MAX_FILE_BYTES: | |
| raise ValueError(f"{source.name} exceeds the per-file limit of {MAX_FILE_BYTES // (1024*1024)} MB.") | |
| extension = source.suffix.lower() | |
| if extension not in ALLOWED_EXTENSIONS: | |
| raise ValueError(f"Unsupported file type for {source.name}: {extension or '<none>'}") | |
| original_name = source.name | |
| safe_name = safe_slug(original_name, f"asset_{index:03d}{extension}") | |
| candidate = f"{index:03d}_{safe_name}" | |
| while candidate in used_names: | |
| candidate = f"{index:03d}_{sha256_file(source)[:8]}_{safe_name}" | |
| used_names.add(candidate) | |
| destination = evidence_dir / candidate | |
| shutil.copyfile(source, destination) | |
| if sha256_file(source) != sha256_file(destination): | |
| raise RuntimeError(f"Source immutability verification failed for {original_name}.") | |
| override = metadata_map.get(original_name, {}) | |
| linked_ids = override.get("linked_event_ids", common.get("linked_event_ids", [])) or [] | |
| linked_ids = [x for x in linked_ids if x in event_ids] | |
| public_status = override.get("public_status", common.get("public_status", "UNKNOWN")) | |
| redaction_state = override.get("redaction_state", common.get("redaction_state", "REDACTION_REQUIRED")) | |
| findings = scan_file(source) | |
| if findings: | |
| warnings.append(f"{original_name}: sensitive-data scanner flagged {', '.join(findings)}.") | |
| if public_status == "PUBLIC" and redaction_state not in {"REDACTED", "BLOCKED_SENSITIVE"}: | |
| redaction_state = "BLOCKED_SENSITIVE" | |
| must_stop = True | |
| if redaction_state == "BLOCKED_SENSITIVE": | |
| must_stop = True | |
| media_type = override.get("media_type") or mimetypes.guess_type(original_name)[0] or "application/octet-stream" | |
| manifests.append({ | |
| "asset_id": f"ASSET_{index:04d}", | |
| "original_filename": original_name, | |
| "sha256": sha256_file(source), | |
| "byte_size": size, | |
| "media_type": media_type, | |
| "source_surface": str(override.get("source_surface", common.get("source_surface", "Creator-provided source"))), | |
| "source_url": str(override.get("source_url", common.get("source_url", ""))), | |
| "creator_description": str(override.get("creator_description", common.get("creator_description", "Creator-provided evidence asset"))), | |
| "date_represented": str(override.get("date_represented", common.get("date_represented", ""))), | |
| "uploaded_at": created_at, | |
| "redaction_state": redaction_state, | |
| "public_status": public_status, | |
| "linked_event_ids": linked_ids, | |
| "execution_blocked": True, | |
| }) | |
| return manifests, warnings, must_stop | |
| def _write_csv(path: Path, rows: list[dict[str, Any]], fieldnames: list[str]) -> None: | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| with path.open("w", encoding="utf-8", newline="") as handle: | |
| writer = csv.DictWriter(handle, fieldnames=fieldnames, lineterminator="\n", extrasaction="ignore") | |
| writer.writeheader() | |
| for row in rows: | |
| rendered = {} | |
| for field in fieldnames: | |
| value = row.get(field, "") | |
| if isinstance(value, (list, dict)): | |
| rendered[field] = json.dumps(value, ensure_ascii=False, sort_keys=True) | |
| elif value is None: | |
| rendered[field] = "" | |
| else: | |
| rendered[field] = value | |
| writer.writerow(rendered) | |
| def _md_list(items: Iterable[str], empty: str = "None recorded.") -> str: | |
| clean = [str(item).strip() for item in items if str(item).strip()] | |
| return "\n".join(f"- {item}" for item in clean) if clean else f"- {empty}" | |
| def _dpio_markdown(read: dict[str, Any]) -> str: | |
| sections = [ | |
| "# DPIO Procedural Read v0.1.0", | |
| "", | |
| f"**Capsule:** `{read['creator_capsule_id']}` ", | |
| f"**Procedure:** `{read['procedure']}` ", | |
| f"**Claim ceiling:** `{read['claim_ceiling']}` ", | |
| "**Human review:** Required", | |
| "", | |
| "## Observed facts", | |
| _md_list(read["observed_facts"]), | |
| "", | |
| "## Creator-reported context", | |
| _md_list(read["creator_reported_context"]), | |
| "", | |
| "## Supported or provisional inferences", | |
| _md_list(read["supported_inferences"]), | |
| "", | |
| "## Unresolved causes", | |
| _md_list(read["unresolved_causes"]), | |
| "", | |
| "## Controls and competing conditions", | |
| _md_list(read["controls_and_competing_conditions"]), | |
| "", | |
| "## Fixed execution order", | |
| _md_list([f"{item['sequence']}. {item['stage']} — {item['state']}" for item in read["execution_order_receipt"]]), | |
| "", | |
| "## Collapsed causal families", | |
| _md_list([f"{item['causal_family_id']}: {item['description']} Members: {', '.join(item['member_event_ids'])}" for item in read["causal_families"]]), | |
| "", | |
| "## Frozen discriminator predictions", | |
| _md_list([f"{item['prediction_id']}: {item['statement']} [{item['state']}]" for item in read["frozen_discriminator_predictions"]]), | |
| "", | |
| "## Competing hypotheses", | |
| ] | |
| for hypothesis in read["competing_hypotheses"]: | |
| sections.extend([ | |
| f"### {hypothesis['hypothesis']}", | |
| f"**Current fit:** `{hypothesis['fit']}`", | |
| "", | |
| "Supporting observations:", | |
| _md_list(hypothesis["supporting_observations"]), | |
| "", | |
| "Falsifiers:", | |
| _md_list(hypothesis["falsifiers"]), | |
| "", | |
| ]) | |
| sections.extend(["## Source-return requests", _md_list(read["source_return_requests"]), ""]) | |
| return "\n".join(sections) | |
| def _causal_arc_markdown(packet: dict[str, Any]) -> str: | |
| return "\n".join([ | |
| "# Bounded Causal Arc Packet v0.1.0", | |
| "", | |
| f"**Packet:** `{packet['packet_id']}` ", | |
| f"**Creator capsule:** `{packet['creator_capsule_id']}` ", | |
| f"**Loop state:** `{packet['loop_state']}` ", | |
| "**Human review:** Required", | |
| "", | |
| "## OBSERVED_FACTS", | |
| _md_list(packet["observed_facts"]), | |
| "", | |
| "## CREATOR_REPORTED_CONTEXT", | |
| _md_list(packet["creator_reported_context"]), | |
| "", | |
| "## SUPPORTED_INFERENCES", | |
| _md_list(packet["supported_inferences"]), | |
| "", | |
| "## UNRESOLVED_CAUSES", | |
| _md_list(packet["unresolved_causes"]), | |
| "", | |
| "## Source-return requests", | |
| _md_list(packet["source_return_requests"]), | |
| "", | |
| "## Claims not to make", | |
| _md_list(packet["claims_not_to_make"]), | |
| "", | |
| ]) | |
| def build_packet(config: dict[str, Any], schema_dir: str | Path, output_root: str | Path | None = None) -> dict[str, Any]: | |
| cleanup_old_workspaces() | |
| created_at = str(config.get("created_at") or utc_now_iso()).strip() | |
| events = parse_events(config.get("events")) | |
| claims = parse_claims(config.get("claims")) | |
| metric_receipts = parse_metric_receipts(config.get("metric_receipts")) | |
| uploads = _normalize_upload_paths(config.get("uploaded_files")) | |
| metadata_map = _asset_metadata_map(str(config.get("asset_metadata_json", ""))) | |
| platform = str(config.get("platform", "")).strip() or "UNSPECIFIED_PLATFORM" | |
| handle = str(config.get("handle_or_pseudonym", "")).strip() or "UNSPECIFIED_CREATOR" | |
| exact_description = str(config.get("creator_exact_description", "")).strip() | |
| if not exact_description: | |
| raise ValueError("The creator's exact content description is required and cannot be system-inferred.") | |
| capsule_id = str(config.get("capsule_id", "")).strip() or capsule_id_from_seed(platform, handle, created_at, exact_description) | |
| consent_state = str(config.get("consent_state", "PENDING")) | |
| consent_scope = str(config.get("consent_scope", "PRIVATE_PACKET_ONLY")) | |
| event_ids = {e["event_id"] for e in events} | |
| if len(event_ids) != len(events): | |
| raise ValueError("Event IDs must be unique.") | |
| workspace_root = Path(output_root) if output_root else Path(tempfile.mkdtemp(prefix="substrate_creator_packet_")) | |
| packet_dir = workspace_root / f"CREATOR_PACKET_{safe_slug(capsule_id)}" | |
| if packet_dir.exists(): | |
| shutil.rmtree(packet_dir) | |
| packet_dir.mkdir(parents=True) | |
| linked_event_ids = parse_lines(str(config.get("asset_linked_event_ids", "")).replace(";", "\n")) | |
| common_asset = { | |
| "source_surface": str(config.get("asset_source_surface", "Creator-provided source")), | |
| "source_url": str(config.get("asset_source_url", "")), | |
| "creator_description": str(config.get("asset_creator_description", "Creator-provided evidence asset")), | |
| "date_represented": str(config.get("asset_date_represented", "")), | |
| "redaction_state": str(config.get("asset_redaction_state", "REDACTION_REQUIRED")), | |
| "public_status": str(config.get("asset_public_status", "UNKNOWN")), | |
| "linked_event_ids": linked_event_ids, | |
| } | |
| assets, sensitive_warnings, must_stop = _copy_and_manifest_assets( | |
| uploads, packet_dir / "evidence", created_at, event_ids, common_asset, metadata_map | |
| ) | |
| text_scan_fields = [ | |
| exact_description, str(config.get("creator_reported_context", "")), | |
| str(config.get("controls", "")), str(config.get("source_return_requests", "")), | |
| str(config.get("account_url", "")), str(config.get("asset_source_url", "")), | |
| ] | |
| text_findings = sorted(set(item for text in text_scan_fields for item in scan_text(text))) | |
| if text_findings: | |
| sensitive_warnings.append("Packet text fields flagged: " + ", ".join(text_findings) + ".") | |
| if "possible_secret_or_token" in text_findings or consent_scope == "PUBLIC_EXHIBIT_CANDIDATE": | |
| must_stop = True | |
| requested_loop_state = str(config.get("loop_state", "ACTIVE_REVIEW")) | |
| closure_blockers = parse_lines(config.get("closure_blockers")) | |
| if must_stop: | |
| loop_state = "MUST_STOP" | |
| closure_blockers.append("Sensitive-data gate requires redaction or removal before further routing.") | |
| elif requested_loop_state == "CLOSED_FOR_CURRENT_SCOPE" and closure_blockers: | |
| loop_state = "FALSE_CLOSURE_RISK" | |
| else: | |
| loop_state = requested_loop_state | |
| consent = { | |
| "consent_state": consent_state, | |
| "scope": consent_scope, | |
| "attribution_name": str(config.get("attribution_name", "")), | |
| "granted_at": created_at if consent_state == "GRANTED" else None, | |
| "withdrawn_at": created_at if consent_state == "WITHDRAWN" else None, | |
| "creator_acknowledgements": DEFAULT_ACKNOWLEDGEMENTS, | |
| "publication_requires_additional_review": True, | |
| } | |
| source_returns = parse_lines(config.get("source_return_requests")) | |
| capsule = { | |
| "capsule_id": capsule_id, | |
| "schema_version": "v0.1.0", | |
| "created_at": created_at, | |
| "creator": { | |
| "display_name": str(config.get("display_name", "")).strip(), | |
| "platform": platform, | |
| "handle_or_pseudonym": handle, | |
| "account_url": str(config.get("account_url", "")).strip(), | |
| "account_created": str(config.get("account_created", "")).strip(), | |
| "followers": int(config["followers"]) if config.get("followers") not in (None, "") else None, | |
| "followers_observed_at": str(config.get("followers_observed_at", "")).strip(), | |
| "attribution_preference": str(config.get("attribution_preference", "PSEUDONYMOUS")), | |
| }, | |
| "consent": consent, | |
| "content_topology": { | |
| "creator_exact_description": exact_description, | |
| "recurring_subjects": parse_lines(config.get("recurring_subjects")), | |
| "criticized_institutions_or_conduct": parse_lines(config.get("criticized_institutions_or_conduct")), | |
| "modes": parse_lines(config.get("modes")), | |
| "formats": parse_lines(config.get("formats")), | |
| "posting_cadence": str(config.get("posting_cadence", "")).strip(), | |
| "subject_change_near_event": str(config.get("subject_change_near_event", "")).strip(), | |
| "system_inferred_labels": [], | |
| }, | |
| "account_baseline": { | |
| "baseline_start": str(config.get("baseline_start", "")).strip(), | |
| "baseline_end": str(config.get("baseline_end", "")).strip(), | |
| "typical_impressions": float(config["typical_impressions"]) if config.get("typical_impressions") not in (None, "") else None, | |
| "typical_engagement_rate": float(config["typical_engagement_rate"]) if config.get("typical_engagement_rate") not in (None, "") else None, | |
| "prior_high_reach_examples": parse_lines(config.get("prior_high_reach_examples")), | |
| "follower_delivery_baseline": str(config.get("follower_delivery_baseline", "")).strip(), | |
| "notification_baseline": str(config.get("notification_baseline", "")).strip(), | |
| "verification_state": str(config.get("verification_state", "")).strip(), | |
| "subscription_state": str(config.get("subscription_state", "")).strip(), | |
| }, | |
| "temporal_events": events, | |
| "evidence_assets": assets, | |
| "claims": claims, | |
| "loop_state": loop_state, | |
| "review_pack": { | |
| "closure_blockers": sorted(set(closure_blockers)), | |
| "claims_not_to_make": CLAIMS_NOT_TO_MAKE, | |
| "evidence_to_reduce_uncertainty": parse_lines(config.get("evidence_to_reduce_uncertainty")), | |
| "repair_deltas": parse_lines(config.get("repair_deltas")), | |
| "next_reviewer_targets": parse_lines(config.get("next_reviewer_targets")), | |
| }, | |
| "parent_capsule_ids": parse_lines(config.get("parent_capsule_ids")), | |
| "correction_history": [], | |
| "source_return_request": source_returns, | |
| } | |
| distribution_report = analyze_distribution( | |
| metric_receipts, | |
| float(config["typical_impressions"]) if config.get("typical_impressions") not in (None, "") else None, | |
| ) | |
| truth_surface_ledger = { | |
| "version": "v0.2.0", | |
| "source_account_truth": metric_receipts, | |
| "creator_reported_truth": parse_lines(config.get("creator_reported_context")), | |
| "system_inference_boundary": [ | |
| "Unknown mechanism cannot erase a documented platform-displayed state.", | |
| "A displayed metric receipt does not independently establish unique-human count, internal counting method, mechanism, authorization, or intent.", | |
| "Truth is downstream from honest preservation of the source account and causal order.", | |
| ], | |
| } | |
| holographic_precondition = { | |
| "version": "v0.2.0", | |
| "capsule_id": capsule_id, | |
| "individual_trace_review_required": True, | |
| "source_bound_receipts_present": bool(assets or metric_receipts), | |
| "temporal_order_present": bool(events), | |
| "uncertainty_preserved": True, | |
| "comparison_consent_scope": consent_scope, | |
| "comparison_ready_now": False, | |
| "human_review_required": True, | |
| "boundary": "No trace may be stacked into a holographic comparison before individual review, consent eligibility, checksum continuity, and false-closure review.", | |
| } | |
| controls = parse_lines(config.get("controls")) | |
| creator_context = parse_lines(config.get("creator_reported_context")) | |
| dpio_read = build_dpio_read(capsule, controls, creator_context) | |
| causal_arc = { | |
| "packet_id": f"CAUSAL_ARC_{capsule_id}", | |
| "creator_capsule_id": capsule_id, | |
| "observed_facts": dpio_read["observed_facts"], | |
| "creator_reported_context": dpio_read["creator_reported_context"], | |
| "supported_inferences": dpio_read["supported_inferences"], | |
| "unresolved_causes": dpio_read["unresolved_causes"], | |
| "competing_hypotheses": dpio_read["competing_hypotheses"], | |
| "source_return_requests": dpio_read["source_return_requests"], | |
| "claims_not_to_make": CLAIMS_NOT_TO_MAKE, | |
| "loop_state": loop_state, | |
| "human_review_required": True, | |
| } | |
| capsule_errors = validate_capsule(capsule, schema_dir) | |
| causal_errors = validate_causal_arc(causal_arc, schema_dir) | |
| all_errors = capsule_errors + causal_errors | |
| if all_errors: | |
| shutil.rmtree(packet_dir, ignore_errors=True) | |
| raise ValueError("Packet validation failed:\n- " + "\n- ".join(all_errors)) | |
| gates = gate_receipt(consent_state, consent_scope) | |
| run_manifest = { | |
| "prototype": "SUBSTRATE_CREATOR_DISTRIBUTION_DPIO_HF_SPACE_PROTOTYPE", | |
| "version": "v0.2.0", | |
| "capsule_id": capsule_id, | |
| "generated_at": created_at, | |
| "runtime_posture": { | |
| "session_local": True, | |
| "external_calls": False, | |
| "database_writes": False, | |
| "automatic_publication": False, | |
| "uploaded_execution": False, | |
| "human_review_required": True, | |
| }, | |
| "gate_receipt": gates, | |
| "sensitive_data_warnings": sensitive_warnings, | |
| "schema_validation": "PASS", | |
| "semantic_validation": "PASS", | |
| "loop_state": loop_state, | |
| "known_limitations": [ | |
| "The prototype does not scrape platforms or verify creator-entered metrics against platform APIs.", | |
| "The prototype preserves platform-displayed receipts but does not infer unique-human counts.", | |
| "The DPIO read is deterministic and rule-bound; it is not an adjudication or legal conclusion.", | |
| "Cross-account comparison is a separate v0.2.0 route requiring at least two eligible packets and explicit human review approval.", | |
| "Image and video uploads are not OCR-scanned for sensitive data; creator redaction remains required.", | |
| ], | |
| } | |
| write_text_lf(packet_dir / "00_READ_ME_FIRST.md", "\n".join([ | |
| "# Creator Evidence Packet", | |
| "", | |
| f"Capsule: `{capsule_id}`", | |
| "", | |
| "This packet preserves creator-provided sources, temporal order, declared uncertainty, and consent boundaries.", | |
| "It does not independently prove suppression, targeting, theft, motive, intent, or executive direction.", | |
| "Human review is required before comparison, publication, or causal attribution.", | |
| "", | |
| f"Current loop state: `{loop_state}`", | |
| ])) | |
| write_canonical_json(packet_dir / "creator_trace_capsule.json", capsule) | |
| write_text_lf(packet_dir / "creator_self_description.md", "# Creator Self-Description\n\n" + exact_description) | |
| write_canonical_json(packet_dir / "evidence_asset_manifest.json", assets) | |
| _write_csv(packet_dir / "evidence_asset_manifest.csv", assets, [ | |
| "asset_id", "original_filename", "sha256", "byte_size", "media_type", "source_surface", | |
| "source_url", "creator_description", "date_represented", "uploaded_at", "redaction_state", | |
| "public_status", "linked_event_ids", "execution_blocked", | |
| ]) | |
| write_canonical_json(packet_dir / "metric_receipts.json", metric_receipts) | |
| _write_csv(packet_dir / "metric_receipts.csv", metric_receipts, METRIC_HEADERS + ["truth_surface", "epistemic_class", "established_fact_boundary"]) | |
| write_canonical_json(packet_dir / "distribution_transition_report.json", distribution_report) | |
| write_text_lf(packet_dir / "distribution_transition_report.md", distribution_report_markdown(distribution_report)) | |
| write_canonical_json(packet_dir / "truth_surface_ledger.json", truth_surface_ledger) | |
| write_canonical_json(packet_dir / "holographic_precondition_receipt.json", holographic_precondition) | |
| write_canonical_json(packet_dir / "source_lineage_receipt.json", { | |
| "prototype_version": "v0.2.0", | |
| "governing_lineage_registry": "lineage/DONOR_REGISTRY.json", | |
| "authority_law": "Authority(State) <= Support(Lineage)", | |
| "receipt_law": "Receipt presence is not semantic validation; source bytes, causal order, and claim boundaries remain separately gated.", | |
| }) | |
| write_canonical_json(packet_dir / "temporal_chain.json", events) | |
| _write_csv(packet_dir / "temporal_chain.csv", events, [ | |
| "event_id", "sequence_index", "event_type", "event_time", "observed_time", "discovered_time", | |
| "recorded_time", "description", "evidence_basis", "predecessor_event_ids", "corrects_event_id", | |
| "state", "uncertainty", | |
| ]) | |
| write_canonical_json(packet_dir / "causal_arc_packet.json", causal_arc) | |
| write_text_lf(packet_dir / "causal_arc_packet.md", _causal_arc_markdown(causal_arc)) | |
| write_canonical_json(packet_dir / "dpio_procedural_read.json", dpio_read) | |
| write_text_lf(packet_dir / "dpio_procedural_read.md", _dpio_markdown(dpio_read)) | |
| write_canonical_json(packet_dir / "dpio_execution_order_receipt.json", dpio_read["execution_order_receipt"]) | |
| write_canonical_json(packet_dir / "causal_families.json", dpio_read["causal_families"]) | |
| write_canonical_json(packet_dir / "frozen_discriminator_predictions.json", dpio_read["frozen_discriminator_predictions"]) | |
| write_canonical_json(packet_dir / "pressure_test_results.json", dpio_read["pressure_test_results"]) | |
| write_canonical_json(packet_dir / "minimum_cut_candidates.json", dpio_read["minimum_cut_candidates"]) | |
| write_text_lf(packet_dir / "source_return_requests.md", "# Source-Return Requests\n\n" + _md_list(source_returns)) | |
| write_text_lf(packet_dir / "claim_boundary.md", "# Claim Boundary\n\n" + _md_list(CLAIMS_NOT_TO_MAKE) + "\n\nL5 attributed cause and L6 intent/motive remain blocked in this prototype.") | |
| write_text_lf(packet_dir / "human_review_checklist.md", "\n".join([ | |
| "# Human Review Checklist", | |
| "", | |
| "- [ ] Verify every observed fact against its cited source.", | |
| "- [ ] Confirm the creator's exact description was preserved without ideological relabeling.", | |
| "- [ ] Check temporal order, correction ancestry, and missing timestamps.", | |
| "- [ ] Test competing hypotheses and their falsifiers.", | |
| "- [ ] Review redaction and third-party privacy.", | |
| "- [ ] Confirm consent scope before any comparison.", | |
| "- [ ] Obtain separate approval before any public exhibit.", | |
| "- [ ] Keep L5/L6 blocked absent supporting sources and explicit human authority.", | |
| ])) | |
| write_canonical_json(packet_dir / "consent_receipt.json", consent) | |
| write_canonical_json(packet_dir / "run_manifest.json", run_manifest) | |
| write_text_lf(packet_dir / "validation_receipt.txt", "\n".join([ | |
| "SCHEMA_VALIDATION: PASS", | |
| "SEMANTIC_VALIDATION: PASS", | |
| "SOURCE_IMMUTABILITY: PASS", | |
| "TEMPORAL_RATCHET: PASS", | |
| "METRIC_RECEIPT_BOUNDARY: PASS", | |
| "HOLOGRAPHIC_PRECONDITION: HELD_FOR_HUMAN_REVIEW", | |
| "CONSENT_GATE: PASS", | |
| "CLAIM_LADDER: PASS", | |
| "FALSE_CLOSURE_GATE: PASS", | |
| f"LOOP_STATE: {loop_state}", | |
| "OVERALL: PASS", | |
| ])) | |
| # Hash every packet member except the checksum ledger itself. | |
| checksums: list[str] = [] | |
| for member in sorted(p for p in packet_dir.rglob("*") if p.is_file() and p.name != "SHA256SUMS.txt"): | |
| checksums.append(f"{sha256_file(member)} {member.relative_to(packet_dir).as_posix()}") | |
| write_text_lf(packet_dir / "SHA256SUMS.txt", "\n".join(checksums)) | |
| zip_path = workspace_root / f"CREATOR_PACKET_{safe_slug(capsule_id)}.zip" | |
| build_deterministic_zip(packet_dir, zip_path) | |
| zip_sha = sha256_file(zip_path) | |
| sidecar_path = workspace_root / f"{zip_path.name}.sha256" | |
| write_text_lf(sidecar_path, f"{zip_sha} {zip_path.name}") | |
| return { | |
| "capsule": capsule, | |
| "causal_arc": causal_arc, | |
| "dpio_read": dpio_read, | |
| "run_manifest": run_manifest, | |
| "packet_dir": str(packet_dir), | |
| "zip_path": str(zip_path), | |
| "sidecar_path": str(sidecar_path), | |
| "zip_sha256": zip_sha, | |
| } | |