data-use-annotate / probe_labels.py
rafmacalaba's picture
annotation review app (per-user queues, Hub-backed rulings, static-safe direct commit)
53ea208 verified
Raw History Blame Contribute Delete
1.59 kB
"""Decision labels for queue items from the singlepass bundle thresholds.
Rule (shared by build_gold_queue.py and rescore_singlepass.py so every
queue files the same vocabulary):
keep score >= per-origin best-F1 threshold (thresholds.json)
drop score <= DROP_FLOOR (confident reject, repo 0.05 edge)
confusion everything between — the human-review zone (was "mid")
unscored score missing (out-of-grid span / not yet rescored)
Per-origin table (published singlepass bundle, best-F1):
fcv_pads_east_africa 0.5 · general_prwp 0.4 · jad_paddy_docs 0.1
jdc_operational 0.6 · refugee_pads 0.5 · reliefweb 0.4
fallback: global_best 0.5
"""
import json
from pathlib import Path
DROP_FLOOR = 0.05
_THR: dict | None = None
def load_thresholds() -> dict:
"""Published singlepass thresholds; mirror of the bundle's thresholds.json."""
global _THR
if _THR is None:
p = (Path(__file__).resolve().parent.parent / "outputs"
/ "gliner-datause-catchall-infer-probe" / "thresholds.json")
_THR = json.loads(p.read_text()) if p.exists() else {}
return _THR
def decide(score, origin, thresholds: dict | None = None) -> str:
if score is None:
return "unscored"
t = thresholds if thresholds is not None else load_thresholds()
thr = (t.get("per_origin") or {}).get(origin, {}).get("thr")
if thr is None:
thr = (t.get("global_best") or {}).get("thr", 0.5)
if score >= thr:
return "keep"
if score <= DROP_FLOOR:
return "drop"
return "confusion"