Spaces:
Running
Running
File size: 2,454 Bytes
0390c03 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 | """IslamicEval-style export of a pipeline result.
The schema mirrors the project's bundled development subset (``data/islamiceval_dev_subset.jsonl``) and the three
subtasks of IslamicEval 2025:
* 1A detection ``Span_Start`` / ``Span_End`` / ``Span_Type`` (``Ayah`` | ``Hadith``; character offsets, end exclusive,
quotation marks excluded)
* 1B verification ``Label`` (``Correct`` | ``Incorrect``)
* 1C correction ``Correction`` the exact source text, or ``NO_SOURCE`` (``خطأ``) when no authentic source exists
Only incorrect spans carry a correction. Nothing is generated: a correction is always the text of a retrieved source.
Check the column names against the organisers' current submission page before an official submission.
"""
from __future__ import annotations
import csv
import io
import json
from typing import Dict, List
NO_SOURCE = "خطأ"
TSV_COLUMNS = ["Response_ID", "Span_Start", "Span_End", "Span_Type", "Label", "Correction"]
def to_benchmark(result: dict, response_id: str = "R001") -> Dict:
rows: List[dict] = []
for span in result["spans"]:
incorrect = span["status"] != "VERIFIED"
correction = None
if incorrect:
proposal = span.get("correction")
correction = proposal["text"] if span["status"] == "CORRECTED" and proposal else NO_SOURCE
rows.append({
"Response_ID": response_id,
"Span_Start": span["start"],
"Span_End": span["end"],
"Span_Type": span["type"],
"Label": "Incorrect" if incorrect else "Correct",
"Correction": correction,
"Status": span["status"], # extra, human-readable: VERIFIED / CORRECTED / UNSUPPORTED / HUMAN_REVIEW
"Text": span["text"],
})
return {"Response_ID": response_id, "Subtask_1A": [r for r in rows], "rows": rows, "summary": result["summary"]}
def to_tsv(benchmark: Dict) -> str:
buffer = io.StringIO()
writer = csv.writer(buffer, delimiter="\t", lineterminator="\n")
writer.writerow(TSV_COLUMNS)
for row in benchmark["rows"]:
writer.writerow([row[c] if row[c] is not None else "" for c in TSV_COLUMNS])
return buffer.getvalue()
def benchmark_json(result: dict, response_id: str = "R001") -> str:
data = to_benchmark(result, response_id)
data["tsv"] = to_tsv(data)
data.pop("Subtask_1A")
return json.dumps(data, ensure_ascii=False)
|