muratcanlaloglu
Publish the v0.2 leadboard.
de8702b
Raw History Blame Contribute Delete
6.97 kB
"""Blind annotation sheets for v0.2+.
export Write a shuffled CSV without expected labels, categories or groups:
dataset/v<version>/annotations/<annotator>.csv
compare Compare a filled sheet with the draft labels: agreement, Cohen's kappa,
disagreements and naturalness flags.
In the sheet, `etiket` takes the option number or the option key. Use `?` when more
than one option fits or none does. Any text in `dogal_degil` flags the Turkish as
unnatural; `not` is free text.
"""
import argparse
import csv
import random
import sys
from collections import Counter
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from benchmark.dataset import VERSIONS, load_cases, load_tasks, version_paths # noqa: E402
COLUMNS = ["no", "id", "soru", "secenekler", "metin", "etiket", "dogal_degil", "not"]
UNSURE = "?"
def sheet_dir(version):
return version_paths(version, "public")[0].parent / "annotations"
def options_text(criteria):
return "\n".join(f"{i}. {key}: {desc}" for i, (key, desc) in enumerate(criteria.items(), 1))
def parse_label(raw, criteria):
raw = raw.strip()
if not raw:
return None
if raw == UNSURE:
return UNSURE
keys = list(criteria)
if raw.isdigit() and 1 <= int(raw) <= len(keys):
return keys[int(raw) - 1]
if raw in criteria:
return raw
raise ValueError(f"unknown label {raw!r}; use 1-{len(keys)}, a key or {UNSURE}")
def read_sheet(path, cases, tasks, ignore_unknown=False):
"""Return {case_id: (label, unnatural, note)} for rows with a label or a flag."""
by_id = {c["id"]: c for c in cases}
out, errors = {}, []
with path.open(encoding="utf-8-sig", newline="") as f:
for row in csv.DictReader(f):
cid = row["id"].strip()
if cid not in by_id:
if not ignore_unknown:
errors.append(f"row {row['no']}: unknown id {cid!r}")
continue
try:
label = parse_label(row["etiket"], tasks[by_id[cid]["task_id"]]["criteria"])
except ValueError as e:
errors.append(f"row {row['no']} ({cid}): {e}")
continue
unnatural = bool(row.get("dogal_degil", "").strip())
note = row.get("not", "").strip()
if label or unnatural or note:
out[cid] = (label, unnatural, note)
return out, errors
def cohen_kappa(pairs):
n = len(pairs)
if not n:
return None
observed = sum(a == b for a, b in pairs) / n
ca, cb = Counter(a for a, _ in pairs), Counter(b for _, b in pairs)
expected = sum(ca[k] * cb[k] for k in ca) / (n * n)
return 1.0 if expected == 1 else (observed - expected) / (1 - expected)
def export(args, cases, tasks):
path = sheet_dir(args.version) / f"{args.annotator}.csv"
if path.exists() and not args.force:
sys.exit(f"{path} already exists; use --force to overwrite (filled labels will be lost)")
rows = list(cases)
random.Random(args.seed).shuffle(rows)
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8-sig", newline="") as f:
writer = csv.writer(f)
writer.writerow(COLUMNS)
for no, c in enumerate(rows, 1):
task = tasks[c["task_id"]]
writer.writerow([no, c["id"], task["instructions"], options_text(task["criteria"]),
c["state"], "", "", ""])
print(f"Wrote {len(rows)} rows to {path}")
def compare(args, cases, tasks):
path = Path(args.file) if args.file else sheet_dir(args.version) / f"{args.annotator}.csv"
sheet, errors = read_sheet(path, cases, tasks)
for e in errors:
print("ERROR:", e)
by_id = {c["id"]: c for c in cases}
labelled = {cid: v for cid, v in sheet.items() if v[0]}
print(f"Labelled {len(labelled)}/{len(cases)} cases in {path.name}")
scored = [(by_id[cid], v) for cid, v in labelled.items() if by_id[cid]["scored"]]
decided = [(c, v) for c, v in scored if v[0] != UNSURE]
pairs = [(c["expected"], v[0]) for c, v in decided]
if pairs:
agree = sum(a == b for a, b in pairs)
print(f"Scored: agreement {agree}/{len(pairs)} ({agree / len(pairs):.0%}), "
f"Cohen's kappa {cohen_kappa(pairs):.2f}, marked '{UNSURE}': {len(scored) - len(decided)}")
ambiguous = [(by_id[cid], v) for cid, v in labelled.items() if not by_id[cid]["scored"]]
if ambiguous:
hits = sum(v[0] == UNSURE or v[0] in c["valid_answers"] for c, v in ambiguous)
print(f"Ambiguous: {hits}/{len(ambiguous)} marked '{UNSURE}' or within valid answers")
disputed = [(c, v) for c, v in scored if v[0] != c["expected"]]
disputed += [(c, v) for c, v in ambiguous if v[0] != UNSURE and v[0] not in c["valid_answers"]]
if disputed:
print("\nDisagreements:")
for c, (label, _, note) in sorted(disputed, key=lambda x: x[0]["id"]):
gold = c["expected"] or "/".join(c["valid_answers"])
print(f" {c['id']} [{c['task_id']}] draft={gold} annotator={label}")
print(f" {c['state']}")
if note:
print(f" not: {note}")
flagged = [(by_id[cid], v) for cid, v in sheet.items() if v[1]]
if flagged:
print("\nFlagged as unnatural:")
for c, (_, _, note) in sorted(flagged, key=lambda x: x[0]["id"]):
print(f" {c['id']}: {c['state']}" + (f"\n not: {note}" if note else ""))
noted = [(by_id[cid], v) for cid, v in sheet.items()
if v[2] and not v[1] and (by_id[cid], v) not in disputed]
if noted:
print("\nOther notes:")
for c, (_, _, note) in sorted(noted, key=lambda x: x[0]["id"]):
print(f" {c['id']}: {note}")
if errors:
sys.exit(1)
def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
sub = parser.add_subparsers(dest="command", required=True)
for name in ("export", "compare"):
p = sub.add_parser(name)
p.add_argument("--version", default="0.2", choices=sorted(VERSIONS))
p.add_argument("--annotator", required=name == "export", help="sheet name, e.g. your first name")
if name == "export":
p.add_argument("--seed", type=int, default=13)
p.add_argument("--force", action="store_true")
else:
p.add_argument("--file", help="filled CSV (default: annotations/<annotator>.csv)")
args = parser.parse_args()
if args.command == "compare" and not (args.file or args.annotator):
parser.error("compare needs --annotator or --file")
public_path, tasks_path = version_paths(args.version, "public")
cases, tasks = load_cases(public_path), load_tasks(tasks_path)
(export if args.command == "export" else compare)(args, cases, tasks)
if __name__ == "__main__":
main()