Spaces:
Running
Running
Download benchmark.py from Ghada-99-Ragab/Islamic3: direct link, hf CLI and curl.
- Browser
- Download file 2.45 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/Islamic3/resolve/main/benchmark.py
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/Islamic3/benchmark.py
-
curl -L -o benchmark.py https://huggingface.co/spaces/Ghada-99-Ragab/Islamic3/resolve/main/benchmark.py
2.45 kB
| """IslamicEval-style export of a pipeline result. | |
| The schema mirrors the project's bundled development subset (``data/islamiceval_dev_subset.jsonl``) and the three | |
| subtasks of IslamicEval 2025: | |
| * 1A detection ``Span_Start`` / ``Span_End`` / ``Span_Type`` (``Ayah`` | ``Hadith``; character offsets, end exclusive, | |
| quotation marks excluded) | |
| * 1B verification ``Label`` (``Correct`` | ``Incorrect``) | |
| * 1C correction ``Correction`` the exact source text, or ``NO_SOURCE`` (``خطأ``) when no authentic source exists | |
| Only incorrect spans carry a correction. Nothing is generated: a correction is always the text of a retrieved source. | |
| Check the column names against the organisers' current submission page before an official submission. | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import io | |
| import json | |
| from typing import Dict, List | |
| NO_SOURCE = "خطأ" | |
| TSV_COLUMNS = ["Response_ID", "Span_Start", "Span_End", "Span_Type", "Label", "Correction"] | |
| def to_benchmark(result: dict, response_id: str = "R001") -> Dict: | |
| rows: List[dict] = [] | |
| for span in result["spans"]: | |
| incorrect = span["status"] != "VERIFIED" | |
| correction = None | |
| if incorrect: | |
| proposal = span.get("correction") | |
| correction = proposal["text"] if span["status"] == "CORRECTED" and proposal else NO_SOURCE | |
| rows.append({ | |
| "Response_ID": response_id, | |
| "Span_Start": span["start"], | |
| "Span_End": span["end"], | |
| "Span_Type": span["type"], | |
| "Label": "Incorrect" if incorrect else "Correct", | |
| "Correction": correction, | |
| "Status": span["status"], # extra, human-readable: VERIFIED / CORRECTED / UNSUPPORTED / HUMAN_REVIEW | |
| "Text": span["text"], | |
| }) | |
| return {"Response_ID": response_id, "Subtask_1A": [r for r in rows], "rows": rows, "summary": result["summary"]} | |
| def to_tsv(benchmark: Dict) -> str: | |
| buffer = io.StringIO() | |
| writer = csv.writer(buffer, delimiter="\t", lineterminator="\n") | |
| writer.writerow(TSV_COLUMNS) | |
| for row in benchmark["rows"]: | |
| writer.writerow([row[c] if row[c] is not None else "" for c in TSV_COLUMNS]) | |
| return buffer.getvalue() | |
| def benchmark_json(result: dict, response_id: str = "R001") -> str: | |
| data = to_benchmark(result, response_id) | |
| data["tsv"] = to_tsv(data) | |
| data.pop("Subtask_1A") | |
| return json.dumps(data, ensure_ascii=False) | |