ONNX
onnxruntime
onnx-mlir
quantization
fp32
ONNX_Models / scripts /validate_netron_exports.py
purejomo's picture
Finalize public ONNX/ONNX-MLIR validation release
ed3aeeb
Raw
History Blame Contribute Delete
17.7 kB
#!/usr/bin/env python3
"""Independently validate Netron source provenance, load evidence, PNG structure, and report coverage."""
from __future__ import annotations
import argparse
import csv
import hashlib
import json
import os
import shlex
import struct
import tempfile
import zlib
from collections import Counter
from pathlib import Path
from typing import Any
EXPECTED_TASK_GROUPS = {
"Anomaly Detection",
"Detection",
"Language Model",
"Segmentation",
"Speech / KWS",
"Vision",
}
def sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def load_csv(path: Path) -> list[dict[str, str]]:
with path.open(newline="", encoding="utf-8") as handle:
return list(csv.DictReader(handle))
def resolve(root: Path, value: str) -> Path:
path = Path(value)
return path if path.is_absolute() else root / path
def command_contains_relocated_path(root: Path, command: str, current: Path) -> bool:
"""Accept a recorded absolute command after moving the repository root."""
current = current.resolve()
try:
relative = current.relative_to(root.resolve())
tokens = shlex.split(command)
except (ValueError, OSError):
return str(current) in command
for token in tokens:
candidate = Path(token)
if candidate.is_absolute():
if candidate == current:
return True
if len(candidate.parts) >= len(relative.parts) and candidate.parts[
-len(relative.parts) :
] == relative.parts:
return True
elif (root / candidate).resolve() == current:
return True
return False
def atomic_json(path: Path, value: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with tempfile.NamedTemporaryFile("w", encoding="utf-8", dir=path.parent, delete=False) as handle:
json.dump(value, handle, indent=2, sort_keys=True, ensure_ascii=False)
handle.write("\n")
temporary = Path(handle.name)
os.replace(temporary, path)
def png_structure(path: Path) -> dict[str, Any]:
"""Validate PNG chunk CRCs without inflating images as large as 500 MP."""
with path.open("rb") as handle:
if handle.read(8) != b"\x89PNG\r\n\x1a\n":
raise ValueError("invalid PNG signature")
width = height = 0
chunk_count = idat_count = 0
seen_ihdr = seen_iend = False
while True:
length_bytes = handle.read(4)
if len(length_bytes) != 4:
raise ValueError("truncated PNG before IEND")
length = struct.unpack(">I", length_bytes)[0]
chunk_type = handle.read(4)
if len(chunk_type) != 4:
raise ValueError("truncated PNG chunk type")
chunk_data = handle.read(length)
crc_bytes = handle.read(4)
if len(chunk_data) != length or len(crc_bytes) != 4:
raise ValueError(f"truncated {chunk_type!r} chunk")
expected_crc = struct.unpack(">I", crc_bytes)[0]
actual_crc = zlib.crc32(chunk_type)
actual_crc = zlib.crc32(chunk_data, actual_crc) & 0xFFFFFFFF
if expected_crc != actual_crc:
raise ValueError(f"CRC mismatch for {chunk_type!r}")
chunk_count += 1
if chunk_type == b"IHDR":
if seen_ihdr or length != 13:
raise ValueError("invalid IHDR")
width, height = struct.unpack(">II", chunk_data[:8])
seen_ihdr = True
elif chunk_type == b"IDAT":
idat_count += 1
elif chunk_type == b"IEND":
if length != 0:
raise ValueError("invalid IEND")
seen_iend = True
if handle.read(1):
raise ValueError("trailing bytes after IEND")
break
if not seen_ihdr or not seen_iend or idat_count == 0 or width <= 0 or height <= 0:
raise ValueError("incomplete PNG structure")
return {"width": width, "height": height, "chunks": chunk_count, "idat_chunks": idat_count}
def check(
checks: list[dict[str, Any]], name: str, ok: bool, target: str = "", detail: Any = None
) -> None:
record: dict[str, Any] = {"check": name, "ok": bool(ok)}
if target:
record["target"] = target
if detail is not None:
record["detail"] = detail
checks.append(record)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--repo-root", type=Path, default=Path.cwd())
parser.add_argument("--report-dir", type=Path, default=Path("reports/graphs/netron"))
parser.add_argument("--output", type=Path, default=Path("reports/graphs/netron/validation.json"))
args = parser.parse_args()
root = args.repo_root.resolve()
report_dir = resolve(root, str(args.report_dir)).resolve()
output = resolve(root, str(args.output)).resolve()
checks: list[dict[str, Any]] = []
rows = load_csv(report_dir / "netron_capture_inventory.csv")
input_rows = load_csv(report_dir / "netron_input_inventory.csv")
model_rows = load_csv(report_dir / "netron_model_matrix.csv")
registry = [row for row in load_csv(root / "model_registry.csv") if row["eligibility"] == "ELIGIBLE"]
summary = json.loads((report_dir / "netron_capture_summary.json").read_text(encoding="utf-8"))
check(checks, "capture_row_count", len(rows) == 42, detail=len(rows))
check(checks, "input_row_count", len(input_rows) == 42, detail=len(input_rows))
check(checks, "model_row_count", len(model_rows) == 21, detail=len(model_rows))
check(checks, "eligible_model_count", len(registry) == 21, detail=len(registry))
keys = {(row["model_id"], row["variant"], row["format"]) for row in rows}
check(checks, "slot_key_unique", len(keys) == 42, detail=len(keys))
check(
checks,
"slot_cartesian_product",
keys
== {
(row["model_id"], variant, fmt)
for row in registry
for variant in ("fp32", "public_quantized")
for fmt in ("onnx",)
},
)
status_counts = Counter(row["capture_status"] for row in rows)
check(checks, "capture_status_counts", status_counts == {"PASS": 42}, detail=dict(status_counts))
format_pass = Counter(row["format"] for row in rows if row["capture_status"] == "PASS")
check(checks, "format_pass_counts", format_pass == {"onnx": 42}, detail=dict(format_pass))
check(checks, "task_groups", {row["task_group"] for row in rows} == EXPECTED_TASK_GROUPS)
canonical = [row for row in rows if row["canonical_s7_selected"].lower() == "true"]
check(checks, "canonical_slot_count", len(canonical) == 42, detail=len(canonical))
check(checks, "canonical_all_pass", all(row["capture_status"] == "PASS" for row in canonical))
check(checks, "canonical_format_counts", Counter(row["format"] for row in canonical) == {"onnx": 42})
input_map = {(row["model_id"], row["variant"], row["format"]): row for row in input_rows}
output_hashes: list[str] = []
for row in rows:
target = f"{row['model_id']}:{row['variant']}:{row['format']}"
key = (row["model_id"], row["variant"], row["format"])
source_row = input_map[key]
check(checks, "capture_input_source_path", row["source_artifact"] == source_row["source_artifact"], target)
check(checks, "capture_input_source_sha", row["current_source_sha256"] == source_row["current_source_sha256"], target)
check(checks, "tool_netron", row["netron_version"] == "9.2.0", target, row["netron_version"])
check(checks, "tool_playwright", row["playwright_version"] == "1.62.0", target, row["playwright_version"])
check(checks, "tool_chromium", row["chromium_version"] == "151.0.7922.34", target, row["chromium_version"])
metadata_path = resolve(root, row["metadata_json"])
check(checks, "metadata_exists", metadata_path.is_file(), target, row["metadata_json"])
if not metadata_path.is_file():
continue
check(checks, "metadata_sha", sha256(metadata_path) == row["metadata_json_sha256"], target)
metadata = json.loads(metadata_path.read_text(encoding="utf-8"))
check(checks, "metadata_identity", (metadata["model_id"], metadata["variant"], metadata["format"]) == key, target)
policy = metadata.get("policy", {})
check(checks, "no_conversion", not policy.get("conversion_performed", False), target)
check(checks, "no_converter_retry", not policy.get("converter_retry_performed", False), target)
check(checks, "no_tflite_to_onnx_for_netron", not policy.get("tflite_to_onnx_for_netron", False), target)
check(checks, "no_model_modification", not policy.get("model_weight_architecture_modified", False), target)
check(checks, "no_allocator_work", not policy.get("allocator_work_performed", False), target)
check(checks, "no_execution_order_inference", not policy.get("execution_order_inferred_from_netron", False), target)
if row["capture_status"] == "NOT_AVAILABLE":
check(checks, "missing_slot_exact", key == ("SP08", "fp32", "tflite"), target)
check(checks, "missing_metadata_status", metadata["status"] == "NOT_AVAILABLE", target)
check(checks, "missing_production_status", metadata["production_evidence"]["stage_status"] == "FAIL", target)
check(checks, "missing_production_failure", metadata["production_evidence"]["failure_code"] == "FAIL_QUANTIZATION_PRESERVATION", target)
check(checks, "missing_pipeline_status", metadata["validation_evidence"]["pipeline_stage_status"] == "FAIL", target)
check(checks, "missing_png_absent", not resolve(root, row["output_png"]).exists(), target)
check(checks, "missing_source_absent", not resolve(root, row["source_artifact"]).exists(), target)
continue
source_path = resolve(root, row["source_artifact"])
check(checks, "source_exists", source_path.is_file(), target, row["source_artifact"])
if source_path.is_file():
current_sha = sha256(source_path)
check(checks, "source_current_sha", current_sha == row["current_source_sha256"], target)
check(checks, "source_recorded_sha", current_sha == row["recorded_source_sha256"], target)
check(checks, "metadata_source_sha", current_sha == metadata["source"]["sha256"], target)
run_result_path = resolve(root, row["source_run_result"])
check(checks, "run_result_exists", run_result_path.is_file(), target)
if run_result_path.is_file():
check(checks, "run_result_sha", sha256(run_result_path) == row["source_run_result_sha256"], target)
check(checks, "metadata_status", metadata["status"] == "PASS" and metadata["failure_code"] is None, target)
check(checks, "capture_method", row["capture_method"] == "NETRON_BROWSER_EXPORT_AS_PNG_CTRL_SHIFT_E", target)
check(checks, "page_http_status", row["page_http_status"] == "200", target, row["page_http_status"])
check(checks, "body_default", "default" in row["body_class"].split(), target, row["body_class"])
check(checks, "origin_nonempty", int(row["origin_child_count"]) > 0, target)
check(checks, "nodes_nonempty", int(row["graph_node_count"]) > 0, target)
check(checks, "page_title_format", row["page_title"].lower().endswith(f".{row['format']}"), target, row["page_title"])
check(checks, "server_exit", row["netron_server_exit_code"] == "0", target, row["netron_server_exit_code"])
check(
checks,
"server_command_source",
command_contains_relocated_path(root, row["netron_server_command"], source_path),
target,
)
check(checks, "server_command_netron", "environment/visualization/netron/.venv/bin/netron" in row["netron_server_command"], target)
server_stderr = resolve(root, metadata["execution"]["server_stderr_log"])
capture_stdout = resolve(root, metadata["execution"]["capture_stdout_log"])
check(checks, "server_stderr_exists", server_stderr.is_file(), target)
check(checks, "capture_stdout_exists", capture_stdout.is_file(), target)
if row.get("reused") == "True":
server_stdout = resolve(root, row["netron_server_stdout_log"])
reuse_marker = server_stdout.read_text(encoding="utf-8", errors="replace") if server_stdout.is_file() else ""
check(checks, "server_loaded_model", "server_not_started=reused_validated_capture" in reuse_marker, target)
check(checks, "server_stopped_cleanly", row["netron_server_exit_code"] == "0", target)
elif server_stderr.is_file():
server_text = server_stderr.read_text(encoding="utf-8", errors="replace")
check(checks, "server_loaded_model", "GET /data/" in server_text and " 200 -" in server_text, target)
check(checks, "server_stopped_cleanly", "Stopping http://127.0.0.1:" in server_text, target)
if capture_stdout.is_file():
check(checks, "capture_log_pass", "status=PASS" in capture_stdout.read_text(encoding="utf-8", errors="replace"), target)
output_png = resolve(root, row["output_png"])
ui_png = resolve(root, row["ui_proof_png"])
check(checks, "netron_filename", output_png.name == f"{row['format']}_netron.png", target)
check(checks, "ui_filename", ui_png.name == f"{row['format']}_netron_ui.png", target)
for label, path, expected_sha, expected_bytes, expected_width, expected_height in (
("netron", output_png, row["output_png_sha256"], row["output_png_bytes"], row["output_png_width"], row["output_png_height"]),
("ui", ui_png, row["ui_proof_png_sha256"], row["ui_proof_png_bytes"], "1920", "1080"),
):
check(checks, f"{label}_png_exists", path.is_file(), target, str(path))
if not path.is_file():
continue
check(checks, f"{label}_png_sha", sha256(path) == expected_sha, target)
check(checks, f"{label}_png_bytes", path.stat().st_size == int(expected_bytes), target)
try:
png = png_structure(path)
check(checks, f"{label}_png_structure", True, target, png)
check(checks, f"{label}_png_dimensions", (png["width"], png["height"]) == (int(expected_width), int(expected_height)), target, png)
except Exception as exception: # noqa: BLE001 - validator reports every artifact failure.
check(checks, f"{label}_png_structure", False, target, str(exception))
if label == "netron":
output_hashes.append(expected_sha)
check(checks, "metadata_export_sha", expected_sha == metadata["netron_export"]["sha256"], target)
check(checks, "metadata_export_dimensions", (int(expected_width), int(expected_height)) == (metadata["netron_export"]["width"], metadata["netron_export"]["height"]), target)
else:
check(checks, "metadata_ui_sha", expected_sha == metadata["ui_proof"]["sha256"], target)
check(checks, "netron_export_hash_count", len(output_hashes) == 42, detail=len(output_hashes))
# Identical visual layouts are allowed (for example closely related vision
# artifacts); source and per-path checksums, not global image uniqueness,
# are the integrity criterion.
check(checks, "model_pairs_all_pass", all(row["pair_netron_status"] == "PASS" for row in model_rows))
check(checks, "summary_status", summary["status"] == "PASS" and summary["failure_code"] is None)
expected_summary = {
"models": 21,
"theoretical_slots": 42,
"source_artifacts_available": 42,
"netron_exports_pass": 42,
"onnx_exports_pass": 42,
"not_available": 0,
"canonical_pair_exports_pass": 42,
"canonical_pair_exports_expected": 42,
"ui_proof_images": 42,
"metadata_records": 42,
}
for key, value in expected_summary.items():
check(checks, f"summary_{key}", summary["counts"][key] == value, detail=summary["counts"][key])
for name in ("README.md", "netron_capture_report.md", "netron_gallery.html", "netron_model_matrix.csv"):
check(checks, "report_file_exists", (report_dir / name).is_file(), name)
report_text = (report_dir / "netron_capture_report.md").read_text(encoding="utf-8")
check(checks, "report_scope", "21개 모델의 FP32·공개 양자화 ONNX 42개" in report_text)
failures = [record for record in checks if not record["ok"]]
result = {
"schema_version": "1.0",
"stage": "T80_NETRON_EXPORT_VALIDATION",
"status": "PASS" if not failures else "FAIL",
"failure_code": None if not failures else "FAIL_ANALYSIS",
"checks_total": len(checks),
"checks_passed": len(checks) - len(failures),
"checks_failed": len(failures),
"checks": checks,
"failures": failures,
"policy": {
"validation_only": True,
"model_files_modified": False,
"allocator_work_performed": False,
},
}
atomic_json(output, result)
print(json.dumps({key: result[key] for key in ("status", "checks_total", "checks_passed", "checks_failed")}, sort_keys=True))
return 0 if not failures else 1
if __name__ == "__main__":
raise SystemExit(main())