File size: 9,468 Bytes
e3584eb
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
import os
import json
import csv
import zipfile
import hashlib
import time
from typing import Dict, Any, List
from benchmark import config

def calculate_sha256(file_path: str) -> str:
    h = hashlib.sha256()
    with open(file_path, "rb") as f:
        while True:
            chunk = f.read(65536)
            if not chunk:
                break
            h.update(chunk)
    return h.hexdigest()

def create_benchmark_package(
    run_id: str,
    source_metadata: Dict[str, Any],
    questions: List[Dict[str, Any]],
    model_pairs_meta: Dict[str, Any],
    plain_models_list: Dict[str, Any],
    rif_models_list: Dict[str, Any],
    raw_logs: List[Dict[str, Any]],
    results_summary: Dict[str, Any]
) -> str:
    """
    Creates a structured benchmark directory, writes JSON/CSV output files,
    generates a checksum manifest, and compresses it into a downloadable zip file.
    Returns: Absolute path of the ZIP package.
    """
    # Create main run root
    run_root_name = f"kalpana-benchmark-{run_id}"
    run_root = os.path.join(config.RUNS_DIR, run_root_name)
    
    # Subdirectories
    subdirs = [
        "source",
        "questions",
        "model-pairs",
        "requests",
        "responses",
        "results",
        "configuration",
        "integrity"
    ]
    for d in subdirs:
        os.makedirs(os.path.join(run_root, d), exist_ok=True)
        
    # 1. README.txt
    readme_content = f"""Kalpanā Independent Multi-Model Benchmark Run Package
Run ID: {run_id}
Timestamp: {time.strftime('%Y-%m-%d %H:%M:%S')}
Label: {results_summary.get('run_label', 'CUSTOM EVALUATOR RUN')}

This package contains all verification details, raw prompts, API logs, and metrics.
No proprietary RIF implementation constants or caching layers are contained here.
"""
    with open(os.path.join(run_root, "README.txt"), "w") as f:
        f.write(readme_content)
        
    # 2. Source artifacts
    source_dir = os.path.join(run_root, "source")
    # If there was an uploaded file, write it (or simulation)
    orig_name = source_metadata.get("filename", "pasted_text.txt")
    with open(os.path.join(source_dir, f"original-upload_{orig_name}"), "w", encoding="utf-8") as f:
        f.write(source_metadata.get("original_text", ""))
        
    with open(os.path.join(source_dir, "extracted-source.txt"), "w", encoding="utf-8") as f:
        f.write(source_metadata.get("extracted_text", ""))
        
    with open(os.path.join(source_dir, "normalized-source.txt"), "w", encoding="utf-8") as f:
        f.write(source_metadata.get("normalized_text", ""))
        
    # If generated synthetic records
    if source_metadata.get("synthetic_records"):
        with open(os.path.join(source_dir, "generated-records.jsonl"), "w") as f:
            for r in source_metadata["synthetic_records"]:
                f.write(json.dumps(r) + "\n")
                
    source_manifest = {
        "filename": orig_name,
        "char_count": len(source_metadata.get("original_text", "")),
        "sha256": hashlib.sha256(source_metadata.get("original_text", "").encode("utf-8")).hexdigest(),
        "source_style": source_metadata.get("style", "custom"),
        "generation_parameters": source_metadata.get("generation_params", {})
    }
    with open(os.path.join(source_dir, "source-manifest.json"), "w") as f:
        json.dump(source_manifest, f, indent=2)
        
    # 3. Questions
    quest_dir = os.path.join(run_root, "questions")
    with open(os.path.join(quest_dir, "questions.jsonl"), "w") as f:
        for q in questions:
            # strip expected answer to build questions list
            q_stripped = q.copy()
            q_stripped.pop("expected_answer", None)
            q_stripped.pop("acceptable_answers", None)
            f.write(json.dumps(q_stripped) + "\n")
            
    with open(os.path.join(quest_dir, "answer-key.jsonl"), "w") as f:
        for q in questions:
            f.write(json.dumps({
                "question_id": q["question_id"],
                "expected_answer": q.get("expected_answer", ""),
                "acceptable_answers": q.get("acceptable_answers", []),
                "supporting_fact_ids": q.get("supporting_fact_ids", [])
            }) + "\n")
            
    # 4. Model Pairs configuration
    mp_dir = os.path.join(run_root, "model-pairs")
    with open(os.path.join(mp_dir, "model-pair-registry.json"), "w") as f:
        json.dump(model_pairs_meta, f, indent=2)
    with open(os.path.join(mp_dir, "plain-api-models.json"), "w") as f:
        json.dump(plain_models_list, f, indent=2)
    with open(os.path.join(mp_dir, "rif-api-models.json"), "w") as f:
        json.dump(rif_models_list, f, indent=2)
        
    # 5. Requests & Responses
    req_dir = os.path.join(run_root, "requests")
    resp_dir = os.path.join(run_root, "responses")
    
    # Sort logs by model family and system
    for log in raw_logs:
        model_name = log.get("model_pair_id", "default_model")
        system_type = log.get("system", "plain")  # "plain" or "rif"
        
        req_file = os.path.join(req_dir, f"{model_name}-{system_type}-requests.jsonl")
        resp_file = os.path.join(resp_dir, f"{model_name}-{system_type}-responses.jsonl")
        
        with open(req_file, "a") as f_req, open(resp_file, "a") as f_resp:
            f_req.write(json.dumps(log.get("raw_request", {})) + "\n")
            f_resp.write(json.dumps(log.get("raw_response", {})) + "\n")
            
    # 6. Results Summary and Paired outcomes
    res_dir = os.path.join(run_root, "results")
    
    # Write CSV for per-question details
    csv_headers = [
        "question_id", "question", "expected", "model_pair",
        "plain_answer", "plain_em", "plain_f1", "plain_latency_ms", "plain_prompt_tokens", "plain_truncated",
        "rif_answer", "rif_em", "rif_f1", "rif_latency_ms", "rif_prompt_tokens", "rif_truncated",
        "source_position", "first_called"
    ]
    
    per_q_results = results_summary.get("per_question", [])
    csv_path = os.path.join(res_dir, "per-question-results.csv")
    with open(csv_path, "w", newline="") as f:
        writer = csv.writer(f)
        writer.writerow(csv_headers)
        for item in per_q_results:
            writer.writerow([
                item.get("question_id"),
                item.get("question"),
                item.get("expected"),
                item.get("model_pair"),
                item.get("plain_answer"),
                item.get("plain_em"),
                item.get("plain_f1"),
                item.get("plain_latency_ms"),
                item.get("plain_prompt_tokens"),
                item.get("plain_truncated"),
                item.get("rif_answer"),
                item.get("rif_em"),
                item.get("rif_f1"),
                item.get("rif_latency_ms"),
                item.get("rif_prompt_tokens"),
                item.get("rif_truncated"),
                item.get("source_position"),
                item.get("first_called")
            ])
            
    # Write summaries
    with open(os.path.join(res_dir, "overall-summary.json"), "w") as f:
        json.dump(results_summary.get("overall", {}), f, indent=2)
        
    # 7. Configurations
    cfg_dir = os.path.join(run_root, "configuration")
    with open(os.path.join(cfg_dir, "benchmark-config.json"), "w") as f:
        json.dump(results_summary.get("config", {}), f, indent=2)
    with open(os.path.join(cfg_dir, "execution-order.json"), "w") as f:
        json.dump(results_summary.get("execution_order", []), f, indent=2)
    with open(os.path.join(cfg_dir, "hardware-comparison.json"), "w") as f:
        json.dump(results_summary.get("hardware_comparison", {}), f, indent=2)
        
    # 8. Create ZIP and calculate checksums
    zip_path = os.path.join(config.RUNS_DIR, f"{run_root_name}.zip")
    
    # We walk the directories and zip everything, excluding integrity files first
    with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zipf:
        for root, dirs, files in os.walk(run_root):
            for file in files:
                file_abs = os.path.join(root, file)
                # Avoid zipping the checksum file if it's already there
                if "sha256-checksums.txt" in file:
                    continue
                arcname = os.path.relpath(file_abs, run_root)
                zipf.write(file_abs, os.path.join(run_root_name, arcname))
                
    # Now generate SHA-256 of the zip and files inside
    checksum_lines = []
    # Add zip file itself
    zip_hash = calculate_sha256(zip_path)
    checksum_lines.append(f"{zip_hash}  {run_root_name}.zip\n")
    
    # Add all files in the directory
    for root, dirs, files in os.walk(run_root):
        for file in files:
            file_abs = os.path.join(root, file)
            if "sha256-checksums.txt" in file:
                continue
            f_hash = calculate_sha256(file_abs)
            rel_path = os.path.relpath(file_abs, run_root)
            checksum_lines.append(f"{f_hash}  {rel_path}\n")
            
    checksum_file_path = os.path.join(run_root, "integrity", "sha256-checksums.txt")
    with open(checksum_file_path, "w") as f:
        f.writelines(checksum_lines)
        
    # Re-zip to include the checksums file
    with zipfile.ZipFile(zip_path, "a") as zipf:
        zipf.write(checksum_file_path, os.path.join(run_root_name, "integrity", "sha256-checksums.txt"))
        
    return zip_path