YOKPilot3 / tools /build_mckf_from_source.py
mfirat007's picture
Upload 17 files
e01a7c3 verified
Raw
History Blame Contribute Delete
31.8 kB
from __future__ import annotations
import argparse
import hashlib
import json
import os
import re
import sys
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
from structural_parser import parse_structural_units
from normative_metadata import (
build_article_normative_metadata,
document_profile,
load_semantic_profiles,
)
from ontology import (
CORPUS_ONTOLOGY_SCHEMA,
DOCUMENT_ONTOLOGY_SCHEMA,
RETRIEVAL_INDEX_SCHEMA,
build_normative_ontology,
)
from rdfox_pilot import pilot_architecture_metadata, write_rdfox_pilot_outputs
from normative_contract import (
assert_valid_corpus,
file_sha256,
validate_rdf_graph,
write_json_atomic,
)
from decision_runtime import NormativeDecisionRuntime
from release_identity import compute_release_manifest
DATA_ROOT = Path(os.getenv("MCKF_BUILD_DATA_DIR", str(ROOT / "data"))).resolve()
REGISTRY_PATH = DATA_ROOT / "registry" / "legal_documents.json"
SEMANTIC_PROFILES_PATH = DATA_ROOT / "registry" / "document_semantic_profiles.json"
ARTICLE_SEMANTIC_CATALOG_PATH = DATA_ROOT / "registry" / "article_semantic_catalog.json"
DECISION_CONTRACT_REGISTRY_PATH = DATA_ROOT / "registry" / "normative_decision_contracts.json"
NORMATIVE_RELATION_REGISTRY_PATH = DATA_ROOT / "registry" / "normative_relations.json"
NORMALIZED_DIR = DATA_ROOT / "normalized"
STRUCTURAL_DIR = DATA_ROOT / "structural"
MCKF_DIR = DATA_ROOT / "mckf"
def write_jsonl(path: Path, rows: list[dict]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8") as file:
for row in rows:
file.write(json.dumps(row, ensure_ascii=False) + "\n")
def read_jsonl(path: Path) -> list[dict[str, Any]]:
if not path.exists():
return []
rows = []
with path.open("r", encoding="utf-8") as file:
for line in file:
line = line.strip()
if line:
rows.append(json.loads(line))
return rows
def infer_roles(text: str, title: str) -> list[str]:
q = (text + " " + title).lower()
roles = []
patterns = {
"amac": ["amac"],
"kapsam": ["kapsam", "kapsar"],
"tanim": ["tanım", "tanımlar", "denir", "ifade eder"],
"gorev": ["görev", "görevleri", "yapar", "yapmak"],
"yetki": ["yetki", "yetkili"],
"sart": ["şart", "koşul", "halinde", "takdirde", "aranır"],
"istisna": ["hariç", "istisna", "ancak", "saklı"],
"sure": ["süre", "yıl", "ay", "gün"],
"atama": ["atanır", "atanması", "seçilir"],
"kurulus": ["kurulur", "açılır", "kapatılır", "teşkil"],
"yaptirim": ["ceza", "disiplin", "kesilir"],
"kurul_olusumu": ["oluşur", "üyeden oluşur", "kişiden oluşur"],
"odeme": ["ücret", "ödenek", "aylık", "gösterge", "tazminat", "katsayı"],
"personel": ["öğretim elemanı", "profesör", "doçent", "araştırma görevlisi", "derece", "kademe"],
"teskilat": ["teşkilat", "fakülte", "enstitü", "yüksekokul", "konservatuvar", "birim"],
"cetvel": ["cetvel", "ek gösterge", "makam tazminatı"],
}
for role, terms in patterns.items():
if any(term in q for term in terms):
roles.append(role)
return roles[:6]
def load_registry() -> list[dict[str, Any]]:
if not REGISTRY_PATH.exists():
raise FileNotFoundError(f"Registry not found: {REGISTRY_PATH}")
documents = json.loads(REGISTRY_PATH.read_text(encoding="utf-8"))
if not isinstance(documents, list):
raise ValueError("legal_documents.json must contain a list")
return documents
def select_documents(document_arg: str | None, build_all: bool) -> list[dict[str, Any]]:
documents = load_registry()
if build_all:
return [doc for doc in documents if doc.get("mckf_status") == "indexed"]
requested = document_arg or "TR-KANUN-2547"
for doc in documents:
if requested in {doc.get("document_id"), doc.get("short_code")}:
return [doc]
raise ValueError(f"Document not found in registry: {requested}")
def build_document(document: dict[str, Any]) -> dict[str, Any]:
document_id = str(document["document_id"])
short_code = str(document["short_code"])
document_title = str(document["title"])
source_path = Path(str(document["source_path"]))
if not source_path.is_absolute():
source_path = DATA_ROOT.parent / source_path
if not source_path.exists():
raise FileNotFoundError(f"Source not found for {document_id}: {source_path}")
text = source_path.read_text(encoding="utf-8")
source_sha256 = hashlib.sha256(text.encode("utf-8")).hexdigest()
versioned_document = dict(document)
versioned_document.update(
{
"source_sha256": source_sha256,
"version_id": f"{document_id}@sha256:{source_sha256[:16]}",
"authority_level": document.get("authority_level") or _authority_level(document.get("document_type", "")),
"validity_status": document.get("validity_status") or "source_snapshot",
}
)
structural = parse_structural_units(text, document_id=document_id)
_attach_document_metadata(structural, versioned_document)
semantic_profiles = load_semantic_profiles(SEMANTIC_PROFILES_PATH)
semantic_profile = document_profile(semantic_profiles, document_id)
NORMALIZED_DIR.mkdir(parents=True, exist_ok=True)
STRUCTURAL_DIR.mkdir(parents=True, exist_ok=True)
MCKF_DIR.mkdir(parents=True, exist_ok=True)
normalized = {
"document_id": document_id,
"document_title": document_title,
"short_code": short_code,
"source_path": document["source_path"],
"article_count": structural["article_count"],
"raw_structural_clauses": structural["clause_count"],
"schedule_count": structural.get("schedule_count", 0),
"source_sha256": source_sha256,
"version_id": versioned_document["version_id"],
}
(NORMALIZED_DIR / f"{short_code}_document.json").write_text(
json.dumps(normalized, ensure_ascii=False, indent=2),
encoding="utf-8",
)
write_jsonl(STRUCTURAL_DIR / f"{short_code}_articles.jsonl", structural["articles"])
write_jsonl(STRUCTURAL_DIR / f"{short_code}_clauses.jsonl", structural["clauses"])
if structural.get("schedules"):
write_jsonl(STRUCTURAL_DIR / f"{short_code}_schedules.jsonl", structural["schedules"])
semantic_units = _semantic_units_from_structural(structural, versioned_document, semantic_profile)
ontology = build_normative_ontology(semantic_units, [], [], document_id=document_id)
ontology["schema"] = DOCUMENT_ONTOLOGY_SCHEMA
ontology["document_title"] = document_title
ontology["short_code"] = short_code
ontology["source_path"] = document["source_path"]
ontology["document_version"] = versioned_document["version_id"]
ontology["source_sha256"] = source_sha256
ontology["generated_at"] = _utc_now()
ontology["provenance"] = {
"source_path": document["source_path"],
"source_sha256": source_sha256,
"builder": "tools/build_mckf_from_source.py",
"extraction_method": "structural_parser+normative_frame_v1",
"semantic_profile": str(SEMANTIC_PROFILES_PATH.relative_to(DATA_ROOT).as_posix()),
}
ontology["architecture_status"] = "rebuilt_from_source"
ontology["runtime_single_source"] = "data/mckf/corpus_mckf_ontology.json"
ontology["freeze_principles"] = {
"single_runtime_source": True,
"runtime_uses_legacy_files": False,
"retrieval_level": "evidence_span",
"llm_role": "synthesize_only_from_selected_evidence",
"source_grounding": "evidence_span",
"document_update_policy": "edit data/source/*.txt and rebuild MCKF outputs; do not edit runtime artifacts manually",
}
ontology.setdefault("stats", {})
ontology["stats"].update(
{
"article_count": structural["article_count"],
"schedule_count": structural.get("schedule_count", 0),
"raw_structural_clauses": structural["clause_count"],
"accepted_mckf_clauses": len(ontology.get("clauses", [])),
"filtered_clause_count": max(0, structural["clause_count"] - len(ontology.get("clauses", []))),
}
)
(MCKF_DIR / f"{short_code}_mckf_ontology.json").write_text(
json.dumps(ontology, ensure_ascii=False, indent=2),
encoding="utf-8",
)
write_jsonl(STRUCTURAL_DIR / f"{short_code}_evidence_spans.jsonl", ontology.get("evidence_spans", []))
semantic_frames = _semantic_frames(ontology.get("evidence_spans", []))
write_jsonl(MCKF_DIR / f"{short_code}_semantic_frames.jsonl", semantic_frames)
(MCKF_DIR / f"{short_code}_retrieval_index.json").write_text(
json.dumps(
{
"schema": RETRIEVAL_INDEX_SCHEMA,
"document_id": document_id,
"document_title": document_title,
"indexes": ontology.get("indexes", {}),
},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
return {
"document": versioned_document,
"ontology": ontology,
"semantic_frames": semantic_frames,
"structural": structural,
}
def build_corpus(results: list[dict[str, Any]]) -> dict[str, Any]:
documents = [result["document"] for result in results]
concepts = _concat(result["ontology"].get("concepts", []) for result in results)
clauses = _concat(result["ontology"].get("clauses", []) for result in results)
evidence_spans = _concat(result["ontology"].get("evidence_spans", []) for result in results)
semantic_frames = _concat(result["semantic_frames"] for result in results)
cross_document_edges = build_cross_document_edges(results)
decision_contracts = _load_decision_contracts()
indexes = _build_corpus_indexes(documents, concepts, clauses, evidence_spans, cross_document_edges)
stats = {
"document_count": len(documents),
"concept_count": len(concepts),
"clause_count": len(clauses),
"evidence_span_count": len(evidence_spans),
"semantic_frame_count": len(semantic_frames),
"cross_document_edge_count": len(cross_document_edges),
"decision_contract_count": len(decision_contracts),
"raw_structural_clauses": sum(result["structural"].get("clause_count", 0) for result in results),
"accepted_mckf_clauses": len(clauses),
"filtered_clause_count": max(0, sum(result["structural"].get("clause_count", 0) for result in results) - len(clauses)),
}
generated_at = _utc_now()
build_id = _corpus_build_id(documents)
institution_ids = sorted(
{str(doc.get("institution_id")) for doc in documents if doc.get("institution_id")}
)
active_institution_id = os.getenv("MCKF_ACTIVE_INSTITUTION_ID", "").strip()
if not active_institution_id and len(institution_ids) == 1:
active_institution_id = institution_ids[0]
corpus = {
"schema": CORPUS_ONTOLOGY_SCHEMA,
"build_id": build_id,
"generated_at": generated_at,
"active_institution_id": active_institution_id,
"institution_ids": institution_ids,
"documents": documents,
"concepts": concepts,
"clauses": clauses,
"evidence_spans": evidence_spans,
"semantic_frames": semantic_frames,
"indexes": indexes,
"cross_document_edges": cross_document_edges,
"decision_contracts": decision_contracts,
"stats": stats,
"pilot_architecture": pilot_architecture_metadata(),
"provenance": {
"builder": "tools/build_mckf_from_source.py",
"generated_at": generated_at,
"document_versions": {doc["document_id"]: doc["version_id"] for doc in documents},
"source_hashes": {doc["document_id"]: doc["source_sha256"] for doc in documents},
},
"freeze_principles": {
"single_runtime_source": True,
"runtime_uses_legacy_files": False,
"retrieval_level": "evidence_span",
"llm_role": "synthesize_only_from_selected_evidence",
"source_grounding": "evidence_span",
"decision_runtime": "deterministic_contract_evaluation",
"llm_may_execute_decisions": False,
},
}
contract_report = assert_valid_corpus(corpus)
write_json_atomic(MCKF_DIR / "corpus_mckf_ontology.json", corpus)
write_jsonl(MCKF_DIR / "corpus_semantic_frames.jsonl", semantic_frames)
write_jsonl(MCKF_DIR / "cross_document_edges.jsonl", cross_document_edges)
(MCKF_DIR / "corpus_retrieval_index.json").write_text(
json.dumps(
{
"schema": "MCKF-HybridRetrievalIndex-v1.0",
"build_id": build_id,
"indexes": indexes,
"stats": stats,
},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
write_rdfox_pilot_outputs(corpus, MCKF_DIR)
rdf_report = validate_rdf_graph(MCKF_DIR / "rdfox_pilot_triples.nt", MCKF_DIR / "mckf_shapes.ttl")
validation_report = dict(contract_report)
validation_report["shacl"] = rdf_report
validation_report["decision_governance"] = NormativeDecisionRuntime(
decision_contracts, build_id
).governance_report()
validation_report["conforms"] = contract_report["conforms"] and rdf_report.get("conforms") is True
write_json_atomic(MCKF_DIR / "validation_report.json", validation_report)
_write_runtime_manifest(build_id, generated_at, validation_report, active_institution_id)
_write_article_semantic_catalog(results, build_id, generated_at)
return corpus
def _write_article_semantic_catalog(
results: list[dict[str, Any]],
build_id: str,
generated_at: str,
) -> None:
documents: dict[str, Any] = {}
article_count = 0
for result in results:
document = result["document"]
articles: dict[str, Any] = {}
by_article_id: dict[str, list[str]] = {}
structural_article_ids = {
str(article.get("article_id", "") or "")
for article in result["structural"].get("articles", []) or []
if article.get("article_id")
}
for concept in result["ontology"].get("concepts", []) or []:
article_id = str(concept.get("article_id", "") or "")
concept_id = str(concept.get("concept_id", "") or "")
metadata = concept.get("normative_metadata", {}) or {}
if article_id in structural_article_ids and concept_id and metadata:
articles[concept_id] = metadata
by_article_id.setdefault(article_id, []).append(concept_id)
article_count += len(articles)
documents[str(document.get("document_id", ""))] = {
"document_title": document.get("title", ""),
"source_sha256": document.get("source_sha256", ""),
"article_count": len(articles),
"articles": articles,
"by_article_id": by_article_id,
}
write_json_atomic(
ARTICLE_SEMANTIC_CATALOG_PATH,
{
"schema": "MCKF-ArticleSemanticCatalog-v1.0",
"build_id": build_id,
"generated_at": generated_at,
"encoding": "UTF-8",
"article_count": article_count,
"documents": documents,
},
)
def build_cross_document_edges(results: list[dict[str, Any]]) -> list[dict[str, Any]]:
seed_payload = (
json.loads(NORMATIVE_RELATION_REGISTRY_PATH.read_text(encoding="utf-8"))
if NORMATIVE_RELATION_REGISTRY_PATH.exists()
else {}
)
seed_edges = list(seed_payload.get("relations", []) or [])
articles = {
(concept.get("document_id"), concept.get("article_id"))
for result in results
for concept in result["ontology"].get("concepts", [])
}
edges = [
edge
for edge in seed_edges
if (edge["source_document_id"], edge["source_article_id"]) in articles
and (edge["target_document_id"], edge["target_article_id"]) in articles
]
seen_ids = {edge["edge_id"] for edge in edges}
for result in results:
source_doc = result["document"].get("document_id")
if source_doc == "TR-KANUN-2547":
continue
for concept in result["ontology"].get("concepts", []):
text = concept.get("source_text", "")
for match in re.finditer(r"2547\s+sayılı.*?(\d+)\s*(?:nci|ncı|üncü|uncu|inci|ıncı)?\s+madd", text, flags=re.IGNORECASE | re.DOTALL):
target_article = f"Madde {match.group(1)}"
edge_id = f"{source_doc.lower().replace('-', '_')}_{concept.get('article_id', '').lower().replace(' ', '_')}_refs_2547_{match.group(1)}"
if edge_id in seen_ids or ("TR-KANUN-2547", target_article) not in articles:
continue
edges.append(
{
"edge_id": edge_id,
"source_document_id": source_doc,
"source_article_id": concept.get("article_id", ""),
"target_document_id": "TR-KANUN-2547",
"target_article_id": target_article,
"relation_type": "references_2547_article",
"description": f"{concept.get('document_title', source_doc)} {concept.get('article_id', '')}, 2547 {target_article} hükmüne atıf yapar.",
}
)
seen_ids.add(edge_id)
return edges
def _attach_document_metadata(structural: dict[str, Any], document: dict[str, Any]) -> None:
document_id = document["document_id"]
document_title = document["title"]
article_titles = {}
for article in structural.get("articles", []):
article["document_title"] = document_title
article["document_short_code"] = document["short_code"]
article_titles[article.get("article_id", "")] = article.get("title", "")
for clause in structural.get("clauses", []):
clause["document_title"] = document_title
clause["document_short_code"] = document["short_code"]
clause["article_title"] = article_titles.get(clause.get("article_id", ""), "")
for schedule in structural.get("schedules", []):
schedule["document_title"] = document_title
schedule["document_short_code"] = document["short_code"]
schedule["document_id"] = document_id
def _semantic_units_from_structural(
structural: dict[str, Any],
document: dict[str, Any],
semantic_profile: dict[str, Any] | None = None,
) -> list[dict[str, Any]]:
units = []
occurrences: dict[str, int] = {}
for article in structural.get("articles", []):
local_id = article["article_id"].replace(" ", "_")
occurrences[local_id] = occurrences.get(local_id, 0) + 1
unit_id = f"{local_id}__occ_{occurrences[local_id]:03d}"
units.append(_semantic_unit(article, document, unit_id=unit_id, semantic_profile=semantic_profile))
for schedule in structural.get("schedules", []):
units.append(_semantic_unit(schedule, document, unit_id=schedule["schedule_id"], semantic_profile=semantic_profile))
return units
def _semantic_unit(
unit: dict[str, Any],
document: dict[str, Any],
unit_id: str,
semantic_profile: dict[str, Any] | None = None,
) -> dict[str, Any]:
article_id = unit.get("article_id", "")
title = unit.get("title", "")
source_text = unit.get("source_text", "")
aliases = [title, article_id, document.get("short_code", ""), document.get("title", "")]
if unit.get("class") == "ScheduleUnit":
aliases.extend(["cetvel", "ek gösterge", "makam tazminatı"])
normative_metadata = build_article_normative_metadata(unit, document, semantic_profile)
# The official article heading and the semantic display label are distinct
# data. Expert routing metadata must never overwrite source provenance.
# UI summaries may use ``display_heading`` from normative_metadata, while
# source panels and direct-text requests retain the parsed official title.
display_title = str(title or normative_metadata.get("display_heading", ""))
aliases.extend(normative_metadata.get("query_aliases", []) or [])
aliases.extend(normative_metadata.get("canonical_concepts", []) or [])
return {
"unit_id": unit_id,
"document_id": document["document_id"],
"document_title": document["title"],
"article_id": article_id,
"article_title": display_title,
"heading_path": unit.get("heading_path", []),
"normative_metadata": normative_metadata,
"source_text": source_text,
"semantic_roles": infer_roles(source_text, title),
"subjects": [],
"responsible_entities": [],
"target_entities": [],
"conditions": [],
"exceptions": [],
"obligations": [],
"permissions": [],
"prohibitions": [],
"deadlines_or_durations": [],
"sanctions": [],
"references": [],
"ai_consumption": {"retrieval_aliases": aliases},
}
def _semantic_frames(evidence_spans: list[dict[str, Any]]) -> list[dict[str, Any]]:
frames = []
for evidence in evidence_spans:
frames.append(
{
"evidence_id": evidence.get("evidence_id"),
"parent_clause_id": evidence.get("parent_clause_id"),
"document_id": evidence.get("document_id"),
"document_title": evidence.get("document_title"),
"article_id": evidence.get("article_id"),
"article_title": evidence.get("article_title"),
"label": evidence.get("label"),
"norm_type": evidence.get("norm_type"),
"semantic_roles": evidence.get("semantic_roles"),
"semantic_frame": evidence.get("semantic_frame"),
"source_span": evidence.get("source_span"),
"source_text": evidence.get("source_text"),
}
)
return frames
def _build_corpus_indexes(
documents: list[dict[str, Any]],
concepts: list[dict[str, Any]],
clauses: list[dict[str, Any]],
evidence_spans: list[dict[str, Any]],
cross_document_edges: list[dict[str, Any]],
) -> dict[str, Any]:
by_document = {doc["document_id"]: [] for doc in documents}
by_document_article: dict[str, list[str]] = {}
by_short_code = {doc["short_code"]: doc["document_id"] for doc in documents}
by_evidence: dict[str, dict[str, str]] = {}
by_role: dict[str, list[str]] = {}
by_cross_document_source: dict[str, list[str]] = {}
by_regulated_topic: dict[str, list[str]] = {}
by_hierarchy_term: dict[str, list[str]] = {}
by_legal_effect_type: dict[str, list[str]] = {}
for concept in concepts:
doc_id = concept.get("document_id", "")
concept_id = concept.get("concept_id", "")
article_id = concept.get("article_id", "")
if doc_id and concept_id:
by_document.setdefault(doc_id, []).append(concept_id)
if doc_id and article_id and concept_id:
by_document_article.setdefault(f"{doc_id}::{article_id}", []).append(concept_id)
metadata = concept.get("normative_metadata", {}) or {}
for topic in (
list(metadata.get("canonical_concepts", []) or [])
+ list(metadata.get("regulated_situations", []) or [])
+ [metadata.get("regulates", "")]
):
key = _index_key(topic)
if key:
by_regulated_topic.setdefault(key, []).append(concept_id)
for heading in metadata.get("heading_path", []) or []:
for value in (heading.get("label", ""), heading.get("title", "")):
key = _index_key(value)
if key:
by_hierarchy_term.setdefault(key, []).append(concept_id)
for effect in metadata.get("legal_effect_types", []) or []:
key = _index_key(effect)
if key:
by_legal_effect_type.setdefault(key, []).append(concept_id)
for evidence in evidence_spans:
evidence_id = evidence.get("evidence_id", "")
if not evidence_id:
continue
by_evidence[evidence_id] = {
"document_id": evidence.get("document_id", ""),
"article_id": evidence.get("article_id", ""),
"parent_clause_id": evidence.get("parent_clause_id", ""),
}
for role in evidence.get("semantic_roles", []) or []:
by_role.setdefault(role, []).append(evidence_id)
for edge in cross_document_edges:
source_key = f"{edge.get('source_document_id')}::{edge.get('source_article_id')}"
by_cross_document_source.setdefault(source_key, []).append(edge.get("edge_id", ""))
return {
"by_document": {key: sorted(set(values)) for key, values in by_document.items()},
"by_short_code": by_short_code,
"by_document_article": {key: sorted(set(values)) for key, values in by_document_article.items()},
"by_evidence": by_evidence,
"by_role": {key: sorted(set(values)) for key, values in by_role.items()},
"by_regulated_topic": {key: sorted(set(values)) for key, values in by_regulated_topic.items()},
"by_hierarchy_term": {key: sorted(set(values)) for key, values in by_hierarchy_term.items()},
"by_legal_effect_type": {key: sorted(set(values)) for key, values in by_legal_effect_type.items()},
"by_cross_document_source": {key: sorted(set(values)) for key, values in by_cross_document_source.items()},
"clause_count": len(clauses),
}
def _concat(groups) -> list[Any]:
out = []
for group in groups:
out.extend(group or [])
return out
def _index_key(value: Any) -> str:
return re.sub(r"\s+", " ", str(value or "").strip().lower())
def _authority_level(document_type: str) -> str:
return {
"law": "statute",
"regulation": "regulation",
"directive": "institutional_directive",
"decision": "administrative_decision",
"guide": "guidance",
"faq": "informational",
}.get(str(document_type or "").lower(), "unspecified")
def _corpus_build_id(documents: list[dict[str, Any]]) -> str:
source_material = "|".join(
json.dumps(doc, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
for doc in sorted(documents, key=lambda item: str(item.get("document_id", "")))
)
builder_files = [
ROOT / "tools" / "build_mckf_from_source.py",
ROOT / "structural_parser.py",
ROOT / "ontology.py",
ROOT / "evidence.py",
ROOT / "normative_roles.py",
ROOT / "normative_metadata.py",
ROOT / "rdfox_pilot.py",
SEMANTIC_PROFILES_PATH,
MCKF_DIR / "mckf_shapes.ttl",
DECISION_CONTRACT_REGISTRY_PATH,
NORMATIVE_RELATION_REGISTRY_PATH,
]
builder_material = "|".join(
f"{path.name}:{file_sha256(path)}" for path in builder_files if path.exists()
)
material = f"{CORPUS_ONTOLOGY_SCHEMA}|{source_material}|{builder_material}"
return "mckf-" + hashlib.sha256(material.encode("utf-8")).hexdigest()[:20]
def _load_decision_contracts() -> list[dict[str, Any]]:
if not DECISION_CONTRACT_REGISTRY_PATH.exists():
return []
payload = json.loads(DECISION_CONTRACT_REGISTRY_PATH.read_text(encoding="utf-8"))
if isinstance(payload, list):
return payload
if isinstance(payload, dict):
return list(payload.get("contracts", []) or [])
raise ValueError("normative_decision_contracts.json nesne veya liste olmalıdır")
def _utc_now() -> str:
return datetime.now(timezone.utc).isoformat()
def _write_runtime_manifest(
build_id: str,
generated_at: str,
validation_report: dict[str, Any],
active_institution_id: str = "",
) -> None:
artifact_names = [
"corpus_mckf_ontology.json",
"corpus_retrieval_index.json",
"corpus_semantic_frames.jsonl",
"cross_document_edges.jsonl",
"rdfox_pilot_triples.nt",
"rdfox_pilot_rules.dlog",
"mckf_shapes.ttl",
"validation_report.json",
]
artifacts = {}
for name in artifact_names:
path = MCKF_DIR / name
if path.exists():
artifacts[name] = {"sha256": file_sha256(path), "bytes": path.stat().st_size}
release = compute_release_manifest(ROOT, build_id)
write_json_atomic(
MCKF_DIR / "runtime_build_manifest.json",
{
"schema": "MCKF-RuntimeBuildManifest-v1.1",
"build_id": build_id,
"release_id": release["release_id"],
"generated_at": generated_at,
"active_institution_id": active_institution_id,
"validation_conforms": bool(validation_report.get("conforms")),
"behavior_artifacts": release["behavior_artifacts"],
"configured_models": release["configured_models"],
"artifacts": artifacts,
},
)
def main() -> int:
parser = argparse.ArgumentParser(description="Build MitrAnlil MCKF outputs from registered source documents")
parser.add_argument("--all", action="store_true", help="Build every indexed document in data/registry/legal_documents.json and corpus outputs")
parser.add_argument("--document", help="Build one document by document_id or short_code")
args = parser.parse_args()
try:
selected_documents = select_documents(args.document, args.all)
results = [build_document(document) for document in selected_documents]
for result in results:
doc = result["document"]
stats = result["ontology"].get("stats", {})
print(f"MCKF outputs rebuilt from {doc['source_path']}")
print(
f"document={doc['document_id']} articles={stats.get('article_count', 0)} "
f"raw_structural_clauses={stats.get('raw_structural_clauses', 0)} "
f"accepted_mckf_clauses={stats.get('accepted_mckf_clauses', 0)} "
f"filtered_clause_count={stats.get('filtered_clause_count', 0)} "
f"evidence_spans={len(result['ontology'].get('evidence_spans', []))}"
)
if args.all:
corpus = build_corpus(results)
stats = corpus.get("stats", {})
print("Corpus MCKF outputs rebuilt")
print(
f"documents={stats.get('document_count', 0)} clauses={stats.get('clause_count', 0)} "
f"evidence_spans={stats.get('evidence_span_count', 0)} cross_document_edges={stats.get('cross_document_edge_count', 0)}"
)
return 0
except Exception as exc:
print(f"FAIL: {exc}")
return 1
if __name__ == "__main__":
raise SystemExit(main())