File size: 2,454 Bytes
c08b36a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
"""IslamicEval-style export of a pipeline result.

The schema mirrors the project's bundled development subset (``data/islamiceval_dev_subset.jsonl``) and the three
subtasks of IslamicEval 2025:

* 1A detection    ``Span_Start`` / ``Span_End`` / ``Span_Type``  (``Ayah`` | ``Hadith``; character offsets, end exclusive,
                  quotation marks excluded)
* 1B verification ``Label``  (``Correct`` | ``Incorrect``)
* 1C correction   ``Correction``  the exact source text, or ``NO_SOURCE`` (``خطأ``) when no authentic source exists

Only incorrect spans carry a correction. Nothing is generated: a correction is always the text of a retrieved source.
Check the column names against the organisers' current submission page before an official submission.
"""
from __future__ import annotations

import csv
import io
import json
from typing import Dict, List

NO_SOURCE = "خطأ"
TSV_COLUMNS = ["Response_ID", "Span_Start", "Span_End", "Span_Type", "Label", "Correction"]


def to_benchmark(result: dict, response_id: str = "R001") -> Dict:
    rows: List[dict] = []
    for span in result["spans"]:
        incorrect = span["status"] != "VERIFIED"
        correction = None
        if incorrect:
            proposal = span.get("correction")
            correction = proposal["text"] if span["status"] == "CORRECTED" and proposal else NO_SOURCE
        rows.append({
            "Response_ID": response_id,
            "Span_Start": span["start"],
            "Span_End": span["end"],
            "Span_Type": span["type"],
            "Label": "Incorrect" if incorrect else "Correct",
            "Correction": correction,
            "Status": span["status"],            # extra, human-readable: VERIFIED / CORRECTED / UNSUPPORTED / HUMAN_REVIEW
            "Text": span["text"],
        })
    return {"Response_ID": response_id, "Subtask_1A": [r for r in rows], "rows": rows, "summary": result["summary"]}


def to_tsv(benchmark: Dict) -> str:
    buffer = io.StringIO()
    writer = csv.writer(buffer, delimiter="\t", lineterminator="\n")
    writer.writerow(TSV_COLUMNS)
    for row in benchmark["rows"]:
        writer.writerow([row[c] if row[c] is not None else "" for c in TSV_COLUMNS])
    return buffer.getvalue()


def benchmark_json(result: dict, response_id: str = "R001") -> str:
    data = to_benchmark(result, response_id)
    data["tsv"] = to_tsv(data)
    data.pop("Subtask_1A")
    return json.dumps(data, ensure_ascii=False)