| from __future__ import annotations
|
|
|
| import hashlib |
| import json |
| from dataclasses import asdict, dataclass |
| from datetime import date |
| from pathlib import Path
|
| from typing import Any
|
|
|
|
|
| CORE_FRAME_FIELDS = { |
| "actor",
|
| "action",
|
| "object",
|
| "condition",
|
| "exception",
|
| "temporal_constraint",
|
| "modality",
|
| "norm_category",
|
| "confidence",
|
| } |
|
|
| DECISION_OPERATORS = {"eq", "ne", "gt", "gte", "lt", "lte", "in", "not_in", "truthy", "falsy"} |
|
|
|
|
| @dataclass(frozen=True)
|
| class ContractIssue:
|
| severity: str
|
| code: str
|
| entity_id: str
|
| message: str
|
|
|
|
|
| def validate_corpus(corpus: dict[str, Any]) -> dict[str, Any]:
|
| """Validate identity, provenance and referential integrity before runtime use."""
|
| issues: list[ContractIssue] = []
|
| documents = corpus.get("documents", []) or []
|
| concepts = corpus.get("concepts", []) or []
|
| clauses = corpus.get("clauses", []) or []
|
| evidence_spans = corpus.get("evidence_spans", []) or [] |
| decision_contracts = corpus.get("decision_contracts", []) or [] |
| relations = corpus.get("cross_document_edges", []) or [] |
|
|
| document_ids = _unique_ids(documents, "document_id", "document", issues)
|
| concept_ids = _unique_ids(concepts, "concept_id", "concept", issues)
|
| clause_ids = _unique_ids(clauses, "clause_id", "clause", issues)
|
| evidence_ids = _unique_ids(evidence_spans, "evidence_id", "evidence", issues) |
| _unique_ids(decision_contracts, "contract_id", "decision_contract", issues) |
|
|
| if not corpus.get("build_id"):
|
| _add(issues, "error", "missing_build_id", "corpus", "Corpus build_id alanı zorunludur.")
|
|
|
| for document in documents:
|
| entity_id = str(document.get("document_id", ""))
|
| for field in ( |
| "title", "short_code", "source_path", "source_sha256", "version_id", |
| "authority_level", "validity_status", "source_authority", "source_snapshot_date", |
| "temporal_coverage", "jurisdiction", "authority_rank", |
| ): |
| if not document.get(field): |
| _add(issues, "error", f"missing_document_{field}", entity_id, f"Belgede {field} alanı yok.") |
| snapshot = str(document.get("source_snapshot_date", "") or "") |
| try: |
| date.fromisoformat(snapshot) |
| except ValueError: |
| _add(issues, "error", "invalid_source_snapshot_date", entity_id, "Belge kaynak anlık görüntü tarihi ISO-8601 olmalıdır.") |
|
|
| reviewed_concepts = 0 |
| answer_ready_concepts = 0 |
| allowed_review_statuses = {"derived_from_structure", "human_reviewed", "expert_approved", "human_rejected"} |
| for concept in concepts: |
| entity_id = str(concept.get("concept_id", ""))
|
| document_id = str(concept.get("document_id", ""))
|
| if document_id not in document_ids:
|
| _add(issues, "error", "unknown_concept_document", entity_id, "Concept bilinmeyen bir belgeye bağlı.")
|
| if not entity_id.startswith(_id_prefix(document_id)):
|
| _add(issues, "error", "unscoped_concept_id", entity_id, "Concept ID belge namespace'i taşımıyor.")
|
| policy = concept.get("source_policy", {}) or {}
|
| if policy.get("source_lock") is not True:
|
| _add(issues, "error", "source_lock_disabled", entity_id, "Concept source_lock=true olmalıdır.")
|
| metadata = concept.get("normative_metadata", {}) or {} |
| for field in ("regulates", "domain_path", "search_text", "provenance"):
|
| if not metadata.get(field):
|
| _add(
|
| issues,
|
| "error",
|
| f"missing_normative_metadata_{field}",
|
| entity_id,
|
| f"Concept normative_metadata.{field} alanını taşımalıdır.",
|
| ) |
| review_status = str(metadata.get("review_status", "") or "") |
| if review_status not in allowed_review_statuses: |
| _add(issues, "error", "invalid_review_status", entity_id, f"Bilinmeyen uzman inceleme durumu: {review_status}.") |
| if review_status in {"human_reviewed", "expert_approved"}: |
| reviewed_concepts += 1 |
| approved_summary = str(metadata.get("approved_summary", "") or "").strip() |
| if approved_summary: |
| if review_status != "expert_approved": |
| _add(issues, "error", "unapproved_canonical_summary", entity_id, "Kanonik kısa sonuç yalnız expert_approved hükümde yayımlanabilir.") |
| source = str(concept.get("source_text", "") or "").casefold() |
| invalid_terms = [] |
| for point in metadata.get("approved_points", []) or []: |
| if not isinstance(point, dict) or not str(point.get("statement", "") or "").strip(): |
| _add(issues, "error", "invalid_approved_point", entity_id, "Onaylı sonuç birimi statement taşımalıdır.") |
| continue |
| for term in point.get("evidence_terms", []) or []: |
| if str(term).casefold() not in source: |
| invalid_terms.append(str(term)) |
| if invalid_terms: |
| _add(issues, "error", "approved_point_source_mismatch", entity_id, f"Onaylı sonuç kanıtı kaynakta yok: {', '.join(invalid_terms[:4])}.") |
| else: |
| answer_ready_concepts += 1 |
|
|
| for clause in clauses:
|
| entity_id = str(clause.get("clause_id", ""))
|
| document_id = str(clause.get("document_id", ""))
|
| parent_id = str(clause.get("parent_concept_id", ""))
|
| if parent_id not in concept_ids:
|
| _add(issues, "error", "orphan_clause", entity_id, "Clause parent concept bulunamadı.")
|
| if not entity_id.startswith(_id_prefix(document_id)):
|
| _add(issues, "error", "unscoped_clause_id", entity_id, "Clause ID belge namespace'i taşımıyor.")
|
|
|
| low_confidence = 0
|
| for evidence in evidence_spans:
|
| entity_id = str(evidence.get("evidence_id", ""))
|
| document_id = str(evidence.get("document_id", ""))
|
| parent_id = str(evidence.get("parent_clause_id", ""))
|
| if parent_id not in clause_ids:
|
| _add(issues, "error", "orphan_evidence", entity_id, "Evidence parent clause bulunamadı.")
|
| if not entity_id.startswith(_id_prefix(document_id)):
|
| _add(issues, "error", "unscoped_evidence_id", entity_id, "Evidence ID belge namespace'i taşımıyor.")
|
| if not str(evidence.get("source_text", "")).strip():
|
| _add(issues, "error", "empty_evidence", entity_id, "Evidence kaynak metni boş.")
|
| span = evidence.get("source_span", {}) or {}
|
| if not isinstance(span.get("char_start"), int) or not isinstance(span.get("char_end"), int):
|
| _add(issues, "error", "invalid_source_span", entity_id, "Evidence kaynak konumu geçersiz.")
|
| frame = evidence.get("semantic_frame", {}) or {}
|
| missing_frame = sorted(CORE_FRAME_FIELDS - set(frame))
|
| if missing_frame:
|
| _add(issues, "error", "incomplete_semantic_frame", entity_id, f"Eksik frame alanları: {', '.join(missing_frame)}")
|
| if float(frame.get("confidence", 0.0) or 0.0) < 0.45:
|
| low_confidence += 1
|
|
|
| if evidence_spans and low_confidence / len(evidence_spans) > 0.45: |
| _add(
|
| issues,
|
| "warning",
|
| "high_low_confidence_ratio",
|
| "corpus",
|
| f"Evidence kayıtlarının {low_confidence}/{len(evidence_spans)} kadarı düşük çıkarım güvenine sahip.",
|
| ) |
|
|
| article_refs = { |
| (str(item.get("document_id", "")), str(item.get("article_id", ""))) |
| for item in [*concepts, *clauses] |
| if item.get("document_id") and item.get("article_id") |
| } |
| for contract in decision_contracts: |
| _validate_decision_contract(contract, document_ids, article_refs, issues) |
|
|
| for relation in relations: |
| relation_id = str(relation.get("edge_id", "") or "") |
| source_scope = (str(relation.get("source_document_id", "")), str(relation.get("source_article_id", ""))) |
| target_scope = (str(relation.get("target_document_id", "")), str(relation.get("target_article_id", ""))) |
| if source_scope not in article_refs or target_scope not in article_refs: |
| _add(issues, "error", "invalid_relation_scope", relation_id, "Belge ilişkisi corpus içinde bulunmayan hükme bağlanıyor.") |
| if relation.get("approved_interpretation") and relation.get("review_status") != "expert_approved": |
| _add(issues, "error", "unapproved_relation_interpretation", relation_id, "İlişki yorumu expert_approved olmalıdır.") |
|
|
| errors = [issue for issue in issues if issue.severity == "error"]
|
| warnings = [issue for issue in issues if issue.severity == "warning"]
|
| return {
|
| "schema": "MCKF-ValidationReport-v1.2", |
| "build_id": corpus.get("build_id", ""),
|
| "conforms": not errors,
|
| "summary": {
|
| "documents": len(documents),
|
| "concepts": len(concepts),
|
| "clauses": len(clauses),
|
| "evidence_spans": len(evidence_spans), |
| "decision_contracts": len(decision_contracts), |
| "reviewed_concepts": reviewed_concepts, |
| "answer_ready_concepts": answer_ready_concepts, |
| "derived_concepts": max(0, len(concepts) - reviewed_concepts), |
| "review_coverage": round(reviewed_concepts / len(concepts), 4) if concepts else 0.0, |
| "errors": len(errors),
|
| "warnings": len(warnings),
|
| "low_confidence_evidence": low_confidence,
|
| },
|
| "issues": [asdict(issue) for issue in issues],
|
| }
|
|
|
|
|
| def assert_valid_corpus(corpus: dict[str, Any]) -> dict[str, Any]:
|
| report = validate_corpus(corpus)
|
| if not report["conforms"]:
|
| preview = "; ".join(issue["code"] for issue in report["issues"][:8])
|
| raise ValueError(f"MCKF contract validation failed: {preview}")
|
| return report
|
|
|
|
|
| def validate_rdf_graph(triples_path: Path, shapes_path: Path) -> dict[str, Any]:
|
| """Run SHACL when the semantic-web dependencies are installed."""
|
| try:
|
| from pyshacl import validate
|
| from rdflib import Graph
|
| except ImportError as exc:
|
| return {
|
| "status": "dependency_missing",
|
| "conforms": False,
|
| "message": str(exc),
|
| }
|
|
|
| data_graph = Graph()
|
| data_graph.parse(triples_path, format="nt")
|
| shapes_graph = Graph()
|
| shapes_graph.parse(shapes_path, format="turtle")
|
| conforms, results_graph, results_text = validate(
|
| data_graph=data_graph,
|
| shacl_graph=shapes_graph,
|
| inference="rdfs",
|
| abort_on_first=False,
|
| allow_infos=True,
|
| allow_warnings=True,
|
| )
|
| return {
|
| "status": "ok" if conforms else "fail",
|
| "conforms": bool(conforms),
|
| "triple_count": len(data_graph),
|
| "result_count": len(results_graph),
|
| "message": str(results_text)[-4000:],
|
| }
|
|
|
|
|
| def file_sha256(path: Path) -> str:
|
| digest = hashlib.sha256()
|
| with path.open("rb") as handle:
|
| for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
| digest.update(chunk)
|
| return digest.hexdigest()
|
|
|
|
|
| def write_json_atomic(path: Path, payload: dict[str, Any]) -> None:
|
| path.parent.mkdir(parents=True, exist_ok=True)
|
| temporary = path.with_suffix(path.suffix + ".tmp")
|
| temporary.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
| temporary.replace(path)
|
|
|
|
|
| def _unique_ids(rows: list[dict[str, Any]], field: str, kind: str, issues: list[ContractIssue]) -> set[str]:
|
| seen: set[str] = set()
|
| for row in rows:
|
| entity_id = str(row.get(field, "") or "")
|
| if not entity_id:
|
| _add(issues, "error", f"missing_{kind}_id", kind, f"{kind} kimliği boş.")
|
| continue
|
| if entity_id in seen:
|
| _add(issues, "error", f"duplicate_{kind}_id", entity_id, f"Aynı {kind} kimliği birden fazla kullanılmış.")
|
| seen.add(entity_id)
|
| return seen
|
|
|
|
|
| def _id_prefix(document_id: str) -> str: |
| import re
|
|
|
| return re.sub(r"\W+", "_", document_id.lower(), flags=re.UNICODE).strip("_") + "__" |
|
|
|
|
| def _validate_decision_contract( |
| contract: dict[str, Any], |
| document_ids: set[str], |
| article_refs: set[tuple[str, str]], |
| issues: list[ContractIssue], |
| ) -> None: |
| contract_id = str(contract.get("contract_id", "") or "") |
| for field in ("version", "title", "decision_type", "input_schema", "rules", "review_status"): |
| if not contract.get(field): |
| _add(issues, "error", f"missing_decision_{field}", contract_id, f"Karar sözleşmesinde {field} alanı yok.") |
| if contract.get("review_status") != "expert_approved": |
| _add(issues, "error", "decision_contract_not_expert_approved", contract_id, "Çalıştırılabilir karar sözleşmesi expert_approved olmalıdır.") |
| if not contract.get("query_aliases") or not contract.get("fact_extractors"): |
| _add(issues, "error", "decision_contract_not_queryable", contract_id, "Karar sözleşmesi sorgu eş anlamları ve olgu çıkarıcıları taşımalıdır.") |
|
|
| source_refs = contract.get("source_refs", []) or [] |
| if not source_refs: |
| _add(issues, "error", "missing_decision_source_refs", contract_id, "Karar sözleşmesi en az bir kaynak hükme bağlanmalıdır.") |
| for source in source_refs: |
| document_id = str(source.get("document_id", "") or "") |
| article_id = str(source.get("article_id", "") or "") |
| if document_id not in document_ids: |
| _add(issues, "error", "unknown_decision_document", contract_id, f"Bilinmeyen karar kaynağı: {document_id}.") |
| if article_id and (document_id, article_id) not in article_refs: |
| _add(issues, "error", "unknown_decision_article", contract_id, f"Kaynak hüküm corpus içinde bulunamadı: {document_id}::{article_id}.") |
|
|
| fields = ((contract.get("input_schema", {}) or {}).get("fields", {}) or {}) |
| if not isinstance(fields, dict) or not fields: |
| _add(issues, "error", "empty_decision_input_schema", contract_id, "Karar sözleşmesinin olay veri şeması boş olamaz.") |
| fields = {} |
|
|
| rules = contract.get("rules", []) or [] |
| rule_ids: set[str] = set() |
| for rule in rules: |
| rule_id = str(rule.get("rule_id", "") or "") |
| if not rule_id: |
| _add(issues, "error", "missing_decision_rule_id", contract_id, "Karar kuralı kimliği boş.") |
| elif rule_id in rule_ids: |
| _add(issues, "error", "duplicate_decision_rule_id", contract_id, f"Kural kimliği tekrar ediyor: {rule_id}.") |
| rule_ids.add(rule_id) |
| if "outcome" not in rule: |
| _add(issues, "error", "missing_decision_outcome", rule_id or contract_id, "Karar kuralında sonuç yok.") |
| if not isinstance(rule.get("priority", 0), int): |
| _add(issues, "error", "invalid_decision_priority", rule_id or contract_id, "Kural önceliği tam sayı olmalıdır.") |
| conditions = rule.get("when", {}) or {} |
| if not isinstance(conditions, dict) or not ({"all", "any"} & set(conditions)): |
| _add(issues, "error", "invalid_decision_conditions", rule_id or contract_id, "Kural koşulları all veya any listesi taşımalıdır.") |
| continue |
| for mode in ("all", "any"): |
| for condition in conditions.get(mode, []) or []: |
| fact = str(condition.get("fact", "") or "") |
| operator = str(condition.get("operator", "eq") or "eq") |
| if fact not in fields: |
| _add(issues, "error", "unknown_decision_fact", rule_id or contract_id, f"Koşul şemada olmayan olguyu kullanıyor: {fact}.") |
| if operator not in DECISION_OPERATORS: |
| _add(issues, "error", "unknown_decision_operator", rule_id or contract_id, f"Desteklenmeyen operatör: {operator}.") |
|
|
| for requirement in contract.get("judgment_requirements", []) or []: |
| fact = str(requirement.get("fact", "") or "") |
| if fact not in fields: |
| _add(issues, "error", "unknown_judgment_fact", contract_id, f"Değerlendirme alanı şemada yok: {fact}.") |
| if not requirement.get("reason"): |
| _add(issues, "error", "missing_judgment_reason", contract_id, "İnsan değerlendirmesi gereksinimi gerekçe taşımalıdır.") |
| _add(issues, "warning", "requires_legal_judgment", contract_id, str(requirement.get("reason", ""))) |
|
|
|
|
| def _add(issues: list[ContractIssue], severity: str, code: str, entity_id: str, message: str) -> None:
|
| issues.append(ContractIssue(severity, code, entity_id, message))
|
|
|