Spaces:
Running on Zero
Running on Zero
Download scripts/annotation.py from muratcanlaloglu/TurkishDecisionBenchmark: direct link, hf CLI and curl.
- Browser
- Download file 6.97 kB
-
https://huggingface.co/spaces/muratcanlaloglu/TurkishDecisionBenchmark/resolve/main/scripts/annotation.py
- Command line
-
hf download hf://spaces/muratcanlaloglu/TurkishDecisionBenchmark/scripts/annotation.py
-
curl -L -o annotation.py https://huggingface.co/spaces/muratcanlaloglu/TurkishDecisionBenchmark/resolve/main/scripts/annotation.py
6.97 kB
| """Blind annotation sheets for v0.2+. | |
| export Write a shuffled CSV without expected labels, categories or groups: | |
| dataset/v<version>/annotations/<annotator>.csv | |
| compare Compare a filled sheet with the draft labels: agreement, Cohen's kappa, | |
| disagreements and naturalness flags. | |
| In the sheet, `etiket` takes the option number or the option key. Use `?` when more | |
| than one option fits or none does. Any text in `dogal_degil` flags the Turkish as | |
| unnatural; `not` is free text. | |
| """ | |
| import argparse | |
| import csv | |
| import random | |
| import sys | |
| from collections import Counter | |
| from pathlib import Path | |
| sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) | |
| from benchmark.dataset import VERSIONS, load_cases, load_tasks, version_paths # noqa: E402 | |
| COLUMNS = ["no", "id", "soru", "secenekler", "metin", "etiket", "dogal_degil", "not"] | |
| UNSURE = "?" | |
| def sheet_dir(version): | |
| return version_paths(version, "public")[0].parent / "annotations" | |
| def options_text(criteria): | |
| return "\n".join(f"{i}. {key}: {desc}" for i, (key, desc) in enumerate(criteria.items(), 1)) | |
| def parse_label(raw, criteria): | |
| raw = raw.strip() | |
| if not raw: | |
| return None | |
| if raw == UNSURE: | |
| return UNSURE | |
| keys = list(criteria) | |
| if raw.isdigit() and 1 <= int(raw) <= len(keys): | |
| return keys[int(raw) - 1] | |
| if raw in criteria: | |
| return raw | |
| raise ValueError(f"unknown label {raw!r}; use 1-{len(keys)}, a key or {UNSURE}") | |
| def read_sheet(path, cases, tasks, ignore_unknown=False): | |
| """Return {case_id: (label, unnatural, note)} for rows with a label or a flag.""" | |
| by_id = {c["id"]: c for c in cases} | |
| out, errors = {}, [] | |
| with path.open(encoding="utf-8-sig", newline="") as f: | |
| for row in csv.DictReader(f): | |
| cid = row["id"].strip() | |
| if cid not in by_id: | |
| if not ignore_unknown: | |
| errors.append(f"row {row['no']}: unknown id {cid!r}") | |
| continue | |
| try: | |
| label = parse_label(row["etiket"], tasks[by_id[cid]["task_id"]]["criteria"]) | |
| except ValueError as e: | |
| errors.append(f"row {row['no']} ({cid}): {e}") | |
| continue | |
| unnatural = bool(row.get("dogal_degil", "").strip()) | |
| note = row.get("not", "").strip() | |
| if label or unnatural or note: | |
| out[cid] = (label, unnatural, note) | |
| return out, errors | |
| def cohen_kappa(pairs): | |
| n = len(pairs) | |
| if not n: | |
| return None | |
| observed = sum(a == b for a, b in pairs) / n | |
| ca, cb = Counter(a for a, _ in pairs), Counter(b for _, b in pairs) | |
| expected = sum(ca[k] * cb[k] for k in ca) / (n * n) | |
| return 1.0 if expected == 1 else (observed - expected) / (1 - expected) | |
| def export(args, cases, tasks): | |
| path = sheet_dir(args.version) / f"{args.annotator}.csv" | |
| if path.exists() and not args.force: | |
| sys.exit(f"{path} already exists; use --force to overwrite (filled labels will be lost)") | |
| rows = list(cases) | |
| random.Random(args.seed).shuffle(rows) | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| with path.open("w", encoding="utf-8-sig", newline="") as f: | |
| writer = csv.writer(f) | |
| writer.writerow(COLUMNS) | |
| for no, c in enumerate(rows, 1): | |
| task = tasks[c["task_id"]] | |
| writer.writerow([no, c["id"], task["instructions"], options_text(task["criteria"]), | |
| c["state"], "", "", ""]) | |
| print(f"Wrote {len(rows)} rows to {path}") | |
| def compare(args, cases, tasks): | |
| path = Path(args.file) if args.file else sheet_dir(args.version) / f"{args.annotator}.csv" | |
| sheet, errors = read_sheet(path, cases, tasks) | |
| for e in errors: | |
| print("ERROR:", e) | |
| by_id = {c["id"]: c for c in cases} | |
| labelled = {cid: v for cid, v in sheet.items() if v[0]} | |
| print(f"Labelled {len(labelled)}/{len(cases)} cases in {path.name}") | |
| scored = [(by_id[cid], v) for cid, v in labelled.items() if by_id[cid]["scored"]] | |
| decided = [(c, v) for c, v in scored if v[0] != UNSURE] | |
| pairs = [(c["expected"], v[0]) for c, v in decided] | |
| if pairs: | |
| agree = sum(a == b for a, b in pairs) | |
| print(f"Scored: agreement {agree}/{len(pairs)} ({agree / len(pairs):.0%}), " | |
| f"Cohen's kappa {cohen_kappa(pairs):.2f}, marked '{UNSURE}': {len(scored) - len(decided)}") | |
| ambiguous = [(by_id[cid], v) for cid, v in labelled.items() if not by_id[cid]["scored"]] | |
| if ambiguous: | |
| hits = sum(v[0] == UNSURE or v[0] in c["valid_answers"] for c, v in ambiguous) | |
| print(f"Ambiguous: {hits}/{len(ambiguous)} marked '{UNSURE}' or within valid answers") | |
| disputed = [(c, v) for c, v in scored if v[0] != c["expected"]] | |
| disputed += [(c, v) for c, v in ambiguous if v[0] != UNSURE and v[0] not in c["valid_answers"]] | |
| if disputed: | |
| print("\nDisagreements:") | |
| for c, (label, _, note) in sorted(disputed, key=lambda x: x[0]["id"]): | |
| gold = c["expected"] or "/".join(c["valid_answers"]) | |
| print(f" {c['id']} [{c['task_id']}] draft={gold} annotator={label}") | |
| print(f" {c['state']}") | |
| if note: | |
| print(f" not: {note}") | |
| flagged = [(by_id[cid], v) for cid, v in sheet.items() if v[1]] | |
| if flagged: | |
| print("\nFlagged as unnatural:") | |
| for c, (_, _, note) in sorted(flagged, key=lambda x: x[0]["id"]): | |
| print(f" {c['id']}: {c['state']}" + (f"\n not: {note}" if note else "")) | |
| noted = [(by_id[cid], v) for cid, v in sheet.items() | |
| if v[2] and not v[1] and (by_id[cid], v) not in disputed] | |
| if noted: | |
| print("\nOther notes:") | |
| for c, (_, _, note) in sorted(noted, key=lambda x: x[0]["id"]): | |
| print(f" {c['id']}: {note}") | |
| if errors: | |
| sys.exit(1) | |
| def main(): | |
| parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) | |
| sub = parser.add_subparsers(dest="command", required=True) | |
| for name in ("export", "compare"): | |
| p = sub.add_parser(name) | |
| p.add_argument("--version", default="0.2", choices=sorted(VERSIONS)) | |
| p.add_argument("--annotator", required=name == "export", help="sheet name, e.g. your first name") | |
| if name == "export": | |
| p.add_argument("--seed", type=int, default=13) | |
| p.add_argument("--force", action="store_true") | |
| else: | |
| p.add_argument("--file", help="filled CSV (default: annotations/<annotator>.csv)") | |
| args = parser.parse_args() | |
| if args.command == "compare" and not (args.file or args.annotator): | |
| parser.error("compare needs --annotator or --file") | |
| public_path, tasks_path = version_paths(args.version, "public") | |
| cases, tasks = load_cases(public_path), load_tasks(tasks_path) | |
| (export if args.command == "export" else compare)(args, cases, tasks) | |
| if __name__ == "__main__": | |
| main() | |