MaduRox's picture
Upload folder using huggingface_hub
e3584eb verified
Raw
History Blame Contribute Delete
9.47 kB
import os
import json
import csv
import zipfile
import hashlib
import time
from typing import Dict, Any, List
from benchmark import config
def calculate_sha256(file_path: str) -> str:
h = hashlib.sha256()
with open(file_path, "rb") as f:
while True:
chunk = f.read(65536)
if not chunk:
break
h.update(chunk)
return h.hexdigest()
def create_benchmark_package(
run_id: str,
source_metadata: Dict[str, Any],
questions: List[Dict[str, Any]],
model_pairs_meta: Dict[str, Any],
plain_models_list: Dict[str, Any],
rif_models_list: Dict[str, Any],
raw_logs: List[Dict[str, Any]],
results_summary: Dict[str, Any]
) -> str:
"""
Creates a structured benchmark directory, writes JSON/CSV output files,
generates a checksum manifest, and compresses it into a downloadable zip file.
Returns: Absolute path of the ZIP package.
"""
# Create main run root
run_root_name = f"kalpana-benchmark-{run_id}"
run_root = os.path.join(config.RUNS_DIR, run_root_name)
# Subdirectories
subdirs = [
"source",
"questions",
"model-pairs",
"requests",
"responses",
"results",
"configuration",
"integrity"
]
for d in subdirs:
os.makedirs(os.path.join(run_root, d), exist_ok=True)
# 1. README.txt
readme_content = f"""Kalpanā Independent Multi-Model Benchmark Run Package
Run ID: {run_id}
Timestamp: {time.strftime('%Y-%m-%d %H:%M:%S')}
Label: {results_summary.get('run_label', 'CUSTOM EVALUATOR RUN')}
This package contains all verification details, raw prompts, API logs, and metrics.
No proprietary RIF implementation constants or caching layers are contained here.
"""
with open(os.path.join(run_root, "README.txt"), "w") as f:
f.write(readme_content)
# 2. Source artifacts
source_dir = os.path.join(run_root, "source")
# If there was an uploaded file, write it (or simulation)
orig_name = source_metadata.get("filename", "pasted_text.txt")
with open(os.path.join(source_dir, f"original-upload_{orig_name}"), "w", encoding="utf-8") as f:
f.write(source_metadata.get("original_text", ""))
with open(os.path.join(source_dir, "extracted-source.txt"), "w", encoding="utf-8") as f:
f.write(source_metadata.get("extracted_text", ""))
with open(os.path.join(source_dir, "normalized-source.txt"), "w", encoding="utf-8") as f:
f.write(source_metadata.get("normalized_text", ""))
# If generated synthetic records
if source_metadata.get("synthetic_records"):
with open(os.path.join(source_dir, "generated-records.jsonl"), "w") as f:
for r in source_metadata["synthetic_records"]:
f.write(json.dumps(r) + "\n")
source_manifest = {
"filename": orig_name,
"char_count": len(source_metadata.get("original_text", "")),
"sha256": hashlib.sha256(source_metadata.get("original_text", "").encode("utf-8")).hexdigest(),
"source_style": source_metadata.get("style", "custom"),
"generation_parameters": source_metadata.get("generation_params", {})
}
with open(os.path.join(source_dir, "source-manifest.json"), "w") as f:
json.dump(source_manifest, f, indent=2)
# 3. Questions
quest_dir = os.path.join(run_root, "questions")
with open(os.path.join(quest_dir, "questions.jsonl"), "w") as f:
for q in questions:
# strip expected answer to build questions list
q_stripped = q.copy()
q_stripped.pop("expected_answer", None)
q_stripped.pop("acceptable_answers", None)
f.write(json.dumps(q_stripped) + "\n")
with open(os.path.join(quest_dir, "answer-key.jsonl"), "w") as f:
for q in questions:
f.write(json.dumps({
"question_id": q["question_id"],
"expected_answer": q.get("expected_answer", ""),
"acceptable_answers": q.get("acceptable_answers", []),
"supporting_fact_ids": q.get("supporting_fact_ids", [])
}) + "\n")
# 4. Model Pairs configuration
mp_dir = os.path.join(run_root, "model-pairs")
with open(os.path.join(mp_dir, "model-pair-registry.json"), "w") as f:
json.dump(model_pairs_meta, f, indent=2)
with open(os.path.join(mp_dir, "plain-api-models.json"), "w") as f:
json.dump(plain_models_list, f, indent=2)
with open(os.path.join(mp_dir, "rif-api-models.json"), "w") as f:
json.dump(rif_models_list, f, indent=2)
# 5. Requests & Responses
req_dir = os.path.join(run_root, "requests")
resp_dir = os.path.join(run_root, "responses")
# Sort logs by model family and system
for log in raw_logs:
model_name = log.get("model_pair_id", "default_model")
system_type = log.get("system", "plain") # "plain" or "rif"
req_file = os.path.join(req_dir, f"{model_name}-{system_type}-requests.jsonl")
resp_file = os.path.join(resp_dir, f"{model_name}-{system_type}-responses.jsonl")
with open(req_file, "a") as f_req, open(resp_file, "a") as f_resp:
f_req.write(json.dumps(log.get("raw_request", {})) + "\n")
f_resp.write(json.dumps(log.get("raw_response", {})) + "\n")
# 6. Results Summary and Paired outcomes
res_dir = os.path.join(run_root, "results")
# Write CSV for per-question details
csv_headers = [
"question_id", "question", "expected", "model_pair",
"plain_answer", "plain_em", "plain_f1", "plain_latency_ms", "plain_prompt_tokens", "plain_truncated",
"rif_answer", "rif_em", "rif_f1", "rif_latency_ms", "rif_prompt_tokens", "rif_truncated",
"source_position", "first_called"
]
per_q_results = results_summary.get("per_question", [])
csv_path = os.path.join(res_dir, "per-question-results.csv")
with open(csv_path, "w", newline="") as f:
writer = csv.writer(f)
writer.writerow(csv_headers)
for item in per_q_results:
writer.writerow([
item.get("question_id"),
item.get("question"),
item.get("expected"),
item.get("model_pair"),
item.get("plain_answer"),
item.get("plain_em"),
item.get("plain_f1"),
item.get("plain_latency_ms"),
item.get("plain_prompt_tokens"),
item.get("plain_truncated"),
item.get("rif_answer"),
item.get("rif_em"),
item.get("rif_f1"),
item.get("rif_latency_ms"),
item.get("rif_prompt_tokens"),
item.get("rif_truncated"),
item.get("source_position"),
item.get("first_called")
])
# Write summaries
with open(os.path.join(res_dir, "overall-summary.json"), "w") as f:
json.dump(results_summary.get("overall", {}), f, indent=2)
# 7. Configurations
cfg_dir = os.path.join(run_root, "configuration")
with open(os.path.join(cfg_dir, "benchmark-config.json"), "w") as f:
json.dump(results_summary.get("config", {}), f, indent=2)
with open(os.path.join(cfg_dir, "execution-order.json"), "w") as f:
json.dump(results_summary.get("execution_order", []), f, indent=2)
with open(os.path.join(cfg_dir, "hardware-comparison.json"), "w") as f:
json.dump(results_summary.get("hardware_comparison", {}), f, indent=2)
# 8. Create ZIP and calculate checksums
zip_path = os.path.join(config.RUNS_DIR, f"{run_root_name}.zip")
# We walk the directories and zip everything, excluding integrity files first
with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zipf:
for root, dirs, files in os.walk(run_root):
for file in files:
file_abs = os.path.join(root, file)
# Avoid zipping the checksum file if it's already there
if "sha256-checksums.txt" in file:
continue
arcname = os.path.relpath(file_abs, run_root)
zipf.write(file_abs, os.path.join(run_root_name, arcname))
# Now generate SHA-256 of the zip and files inside
checksum_lines = []
# Add zip file itself
zip_hash = calculate_sha256(zip_path)
checksum_lines.append(f"{zip_hash} {run_root_name}.zip\n")
# Add all files in the directory
for root, dirs, files in os.walk(run_root):
for file in files:
file_abs = os.path.join(root, file)
if "sha256-checksums.txt" in file:
continue
f_hash = calculate_sha256(file_abs)
rel_path = os.path.relpath(file_abs, run_root)
checksum_lines.append(f"{f_hash} {rel_path}\n")
checksum_file_path = os.path.join(run_root, "integrity", "sha256-checksums.txt")
with open(checksum_file_path, "w") as f:
f.writelines(checksum_lines)
# Re-zip to include the checksums file
with zipfile.ZipFile(zip_path, "a") as zipf:
zipf.write(checksum_file_path, os.path.join(run_root_name, "integrity", "sha256-checksums.txt"))
return zip_path