"""IslamicEval-style export of a pipeline result. The schema mirrors the project's bundled development subset (``data/islamiceval_dev_subset.jsonl``) and the three subtasks of IslamicEval 2025: * 1A detection ``Span_Start`` / ``Span_End`` / ``Span_Type`` (``Ayah`` | ``Hadith``; character offsets, end exclusive, quotation marks excluded) * 1B verification ``Label`` (``Correct`` | ``Incorrect``) * 1C correction ``Correction`` the exact source text, or ``NO_SOURCE`` (``خطأ``) when no authentic source exists Only incorrect spans carry a correction. Nothing is generated: a correction is always the text of a retrieved source. Check the column names against the organisers' current submission page before an official submission. """ from __future__ import annotations import csv import io import json from typing import Dict, List NO_SOURCE = "خطأ" TSV_COLUMNS = ["Response_ID", "Span_Start", "Span_End", "Span_Type", "Label", "Correction"] def to_benchmark(result: dict, response_id: str = "R001") -> Dict: rows: List[dict] = [] for span in result["spans"]: incorrect = span["status"] != "VERIFIED" correction = None if incorrect: proposal = span.get("correction") correction = proposal["text"] if span["status"] == "CORRECTED" and proposal else NO_SOURCE rows.append({ "Response_ID": response_id, "Span_Start": span["start"], "Span_End": span["end"], "Span_Type": span["type"], "Label": "Incorrect" if incorrect else "Correct", "Correction": correction, "Status": span["status"], # extra, human-readable: VERIFIED / CORRECTED / UNSUPPORTED / HUMAN_REVIEW "Text": span["text"], }) return {"Response_ID": response_id, "Subtask_1A": [r for r in rows], "rows": rows, "summary": result["summary"]} def to_tsv(benchmark: Dict) -> str: buffer = io.StringIO() writer = csv.writer(buffer, delimiter="\t", lineterminator="\n") writer.writerow(TSV_COLUMNS) for row in benchmark["rows"]: writer.writerow([row[c] if row[c] is not None else "" for c in TSV_COLUMNS]) return buffer.getvalue() def benchmark_json(result: dict, response_id: str = "R001") -> str: data = to_benchmark(result, response_id) data["tsv"] = to_tsv(data) data.pop("Subtask_1A") return json.dumps(data, ensure_ascii=False)