#!/usr/bin/env python """Gradio site for human evaluation of object-centric annotations. Run: python app.py [--port 7860] [--share] Responses are appended to responses/responses.csv. """ import argparse import csv import json import os import re from datetime import datetime from pathlib import Path import gradio as gr from filelock import FileLock HERE = Path(__file__).resolve().parent os.environ.setdefault("GRADIO_TEMP_DIR", str(HERE / ".gradio_cache")) MEDIA = HERE / "media" RESP_DIR = HERE / "responses" RESP_CSV = RESP_DIR / "responses.csv" LOCK = HERE / ".responses.csv.lock" CHOICES = ["Correct", "Partially correct", "Incorrect", "Can't tell"] # (column name, question shown to the annotator) DIMENSIONS = [ ("identity", "1. Identity — does `category` + `looks like` describe the object inside the red box?"), ("localization", "2. Localization — do the red box / green mask cover the right object (not a neighbour, hand, or background)?"), ("state_history", "3. Location & state history — is the where / relation / timing (`now`, `history`) correct?"), ("moves", "4. Moves — are the recorded movement events (from → to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"), ("sound_speech", "5. Sound & speech — do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"), ("gaze", "6. Gaze — is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"), ] # ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ---- DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses SCHEDULER = None if DATASET_REPO and os.environ.get("HF_TOKEN"): from huggingface_hub import CommitScheduler, HfApi, hf_hub_download RESP_DIR.mkdir(exist_ok=True) try: # resume from the copy in the dataset api = HfApi(token=os.environ["HF_TOKEN"]) api.create_repo(DATASET_REPO, repo_type="dataset", private=True, exist_ok=True) except Exception as e: print("could not reach the responses dataset:", type(e).__name__, str(e)[:120]) for _v in ("A", "B", "C"): # one file per questionnaire version try: cached = hf_hub_download(DATASET_REPO, f"responses_{_v}.csv", repo_type="dataset", token=os.environ["HF_TOKEN"]) (RESP_DIR / f"responses_{_v}.csv").write_bytes(Path(cached).read_bytes()) print(f"restored version {_v}: {sum(1 for _ in open(cached)) - 1} responses") except Exception as e: # first run: nothing to restore print(f"version {_v}: nothing restored ({type(e).__name__})") SCHEDULER = CommitScheduler(repo_id=DATASET_REPO, repo_type="dataset", folder_path=RESP_DIR, path_in_repo=".", every=2, private=True, token=os.environ["HF_TOKEN"], allow_patterns=["*.csv"]) ITEMS = json.load(open(MEDIA / "index.json")) # Narrations are judged one by one. An object has between 0 and NARR_MAX of them, so the # page holds NARR_MAX radio groups and shows only as many as the current object needs. NARR_MAX = max(1, max(len(it["events"]["narration"]) for it in ITEMS)) COLUMNS = (["timestamp", "annotator", "version", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS] + ["narration_ids"] + [f"narration_{k + 1}" for k in range(NARR_MAX)] + ["comment"]) NO_NARR_Q = ("No narration is linked to this object. Is that right — is there really no narration in this clip " "about it? (All narrations of the clip are listed in the cross-check panel.)") # ---- three shorter questionnaires -------------------------------------------------------- # 95 objects is too long for one sitting, so the clips are dealt into versions A / B / C # (whole clips, so an annotator judges every object of the clips they watch; balanced by # object count and spread over the videos). Each version writes its own CSV. VERSIONS = ["A", "B", "C"] def _split_versions(): clips = {} for i, it in enumerate(ITEMS): clips.setdefault((it["video_id"], it["chunk"]), []).append(i) load = {v: 0 for v in VERSIONS} out = {v: [] for v in VERSIONS} by_video = {} for key in clips: by_video.setdefault(key[0], []).append(key) for vid in sorted(by_video): # within a video, biggest clip first -> lightest version taken = set() for key in sorted(by_video[vid], key=lambda k: (-len(clips[k]), k)): cand = [v for v in VERSIONS if v not in taken] or VERSIONS v = min(cand, key=lambda x: (load[x], x)) taken.add(v) out[v] += clips[key] load[v] += len(clips[key]) return {v: sorted(ix) for v, ix in out.items()} VERSION_ITEMS = _split_versions() def resp_csv(version): return RESP_DIR / f"responses_{version}.csv" # A responses file written with a different column set (older form) is kept under another # name instead of being appended to with misaligned columns. for _p in [RESP_CSV] + [resp_csv(v) for v in VERSIONS]: if _p.exists(): with open(_p, newline="") as _f: _hdr = next(csv.reader(_f), None) if _hdr != COLUMNS or _p == RESP_CSV: _p.rename(RESP_DIR / f"{_p.stem}_old_{datetime.now():%Y%m%d_%H%M%S}.csv") def hms(t, dec=1): """Seconds of the full recording -> H:MM:SS.s (the unit shown everywhere on the site).""" if t is None: return "?" t = round(float(t), dec) h, m = int(t // 3600), int(t % 3600 // 60) s = t - h * 3600 - m * 60 return f"{h}:{m:02d}:{s:0{3 + dec if dec else 2}.{dec}f}" def place(anchor): """`P02_counter.002` -> `the counter #2`; `person_01` -> `the wearer's hand`.""" if not anchor or anchor == "unknown": return "an unknown place" if anchor.startswith("person"): return "the wearer's hand" kind, _, num = re.sub(r"^P\d\d_", "", anchor).partition(".") return f"the {kind.replace('_', ' ')}" + (f" #{int(num)}" if num.isdigit() else "") def where(rel, anchor): if rel == "held_by": return "held in the wearer's hand" return {"inside": "inside", "on": "on", "hanging_on": "hanging on"}.get(rel, "at") + " " + place(anchor) def state_sentence(st): a, b = st["interval_sec"] w = where(st["relation"], st["anchor"]) if b is None: return f"from {hms(a)} until the end of the clip: {w}" if abs(b - a) < 0.05: return f"at {hms(a)}: {w}" return f"from {hms(a)} to {hms(b)}: {w}" def move_sentence(tr): a, b = tr["interval_sec"] when = f"at {hms(a)}" if abs(b - a) < 0.05 else f"between {hms(a)} and {hms(b)}" return (f"{when} it goes from “{where(tr['from']['relation'], tr['from']['anchor'])}” to " f"“{where(tr['to']['relation'], tr['to']['anchor'])}” (action: {tr['event'].replace('_', ' ')})") def gaze_summary(it): a, att = it["annotation"], it.get("attention") or {} if a.get("looked_at"): gap = att.get("prime_gap_sec") return (f"Yes — first look at {hms(att.get('first_gaze_sec'))}" + (f", about {gap:.1f}s before the hand touches it" if isinstance(gap, (int, float)) else "")) return "No — the memory found no look at this object before the hand touches it" def _join(xs): return "; ".join(f"({i + 1}) {x}" for i, x in enumerate(xs)) def _bul(xs): return "\n".join(f"- {x}" for x in xs) def state_md(st): a, b = st["interval_sec"] w = f"**{where(st['relation'], st['anchor'])}**" if b is None: return f"{hms(a)} → end of clip: {w}" return f"at {hms(a)}: {w}" if abs(b - a) < 0.05 else f"{hms(a)} → {hms(b)}: {w}" def move_md(tr): a, b = tr["interval_sec"] when = f"at **{hms(a)}**" if abs(b - a) < 0.05 else f"**{hms(a)} → {hms(b)}**" return (f"{when}: {where(tr['from']['relation'], tr['from']['anchor'])} **→** " f"{where(tr['to']['relation'], tr['to']['anchor'])} ({tr['event'].replace('_', ' ')})") def dim_questions(it): """Markdown text of the six fixed questions for THIS object: short, claim in bold.""" a, ev, att = it["annotation"], it.get("events") or {}, it.get("attention") or {} obj = f"**{a.get('category')}**" q = {} q["identity"] = f"**1. Identity** — is the boxed object a {obj} that looks like “**{a.get('looks_like')}**”?" q["localization"] = f"**2. Box / mask** — does the **green mask** follow the {obj} **through the video**, and do the **red box** and mask cover it in the **keyframes (zoom on the right)**? (not a neighbour, the hand or background)" states = it.get("states") or [] q["state_history"] = (f"**3. Location** — is the {obj} where the memory says, at these times?\n" + _bul([state_md(x) for x in states] or (a.get("history") or []))) moves = it.get("transitions") or [] q["moves"] = ((f"**4. Moves** *(move = picked up / put down / put into / taken out)* — are these **right**, " f"and is **none missing**?\n" + _bul([move_md(x) for x in moves])) if moves else f"**4. Moves** *(move = picked up / put down / put into / taken out)* — the memory records " f"**no move**. Does the {obj} really **stay in place** for the whole clip?") snd = [f"**{e['label']}** at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("sound") or []] sp = [f"“{e['label']}” at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("speech") or []] q["sound_speech"] = ((f"**5. Sounds 🔊 (turn sound on)** — can you **hear** each one at that time, and is it made " f"**by / with the** {obj}?\n" + _bul(snd)) if snd else f"**5. Sounds 🔊 (turn sound on)** — the memory links **no sound**. Does the {obj} really " "make **no sound** in the clip?") if sp: q["sound_speech"] += "\n\nSpeech in the clip — transcribed correctly?\n" + _bul(sp) HEAD = ("**6. Gaze right before touching** — only the **moment right before the hand touches the object** " "counts, **not the whole clip** (the red dot is on screen all the time).\n\n") fg, gap = att.get("first_gaze_sec"), att.get("prime_gap_sec") t0, t1 = it["window_sec"] if not a.get("looked_at"): q["gaze"] = (HEAD + f"Memory: the wearer did **NOT look** at the {obj} before touching it. " "In the **2–3 s before the hand touches it**, does the red dot **stay off** the object?") elif not (t0 <= (fg or -1) <= t1): q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **{hms(fg)}**, which is **outside this clip** " f"({hms(t0, 0)}–{hms(t1, 0)}) → please answer **Can't tell**.") else: q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **≈ {hms(fg)}**" + (f" (**{gap:.1f}s before touching it**)" if isinstance(gap, (int, float)) else "") + ". **At that moment**, is the **red dot on (or right next to) the object**?") return [q[k] for k, _ in DIMENSIONS] def narrations_of(it): return sorted(it["events"]["narration"], key=lambda e: e["t0"]) def narration_updates(it): """For each of the NARR_MAX slots: (markdown update, radio update) for this object.""" ns = narrations_of(it) mds, rds = [], [] for k in range(NARR_MAX): if k < len(ns): e = ns[k] mds.append(gr.update(visible=True, value=f"**Narration {k + 1} / {len(ns)}** · **{hms(e['t0'])}–{hms(e['t1'])}** · " f"“{e['label']}” — about **this object**, at the **right time**?")) rds.append(gr.update(visible=True, value=None)) elif k == 0: mds.append(gr.update(visible=True, value="**No narration is linked** to this object. Is that right — does **no " "narration in this clip** talk about it? (all narrations: cross-check panel)")) rds.append(gr.update(visible=True, value=None)) else: mds.append(gr.update(visible=False, value="")) rds.append(gr.update(visible=False, value=None)) return mds + rds N = len(ITEMS) def item_key(it): return (it["video_id"], it["chunk"], it["object_id"]) # ---------------------------------------------------------------- csv i/o def load_done(annotator, version): done = set() p = resp_csv(version) if p.exists(): with open(p, newline="") as f: for r in csv.DictReader(f): if r["annotator"] == annotator: done.add((r["video_id"], r["chunk"], r["object_id"])) return done def _write(row, version): p = resp_csv(version) new = not p.exists() with open(p, "a", newline="") as f: w = csv.DictWriter(f, fieldnames=COLUMNS) if new: w.writeheader() w.writerow(row) def append_row(row, version): RESP_DIR.mkdir(exist_ok=True) with FileLock(str(LOCK)): if SCHEDULER is not None: with SCHEDULER.lock: _write(row, version) else: _write(row, version) if SCHEDULER is not None: # Push right away instead of waiting for the 2-minute tick: a restart of the Space # (every redeploy causes one) would otherwise lose whatever was not yet synced. try: SCHEDULER.trigger() except Exception as e: print("immediate sync failed:", type(e).__name__, str(e)[:120]) # ---------------------------------------------------------------- rendering def fmt_list(xs): return "\n".join(f"- {x}" for x in xs) if xs else "- *(none)*" def fmt_iv(evs): return "\n".join(f"- **{hms(e['t0'])}–{hms(e['t1'])}** {e['label']}" for e in evs) if evs else "- *(none)*" def annotation_md(it): a = it["annotation"] now = a.get("now") or {} lc = it.get("lifecycle") or {} ev = it.get("events") or {} att = it.get("attention") or {} md = f"""## `{it['object_id']}` **video** `{it['video_id']}` / {it['chunk']} window {hms(it['window_sec'][0])}–{hms(it['window_sec'][1])} | field | value | |---|---| | **category** | {a.get('category')} | | **looks like** | {a.get('looks_like')} | | **color / material** | {a.get('color', '?')} / {a.get('material', '?')} | | **at the end of the clip** | {where(now.get('relation'), now.get('at'))} (since {hms(now.get('since_sec'))}){(' · contains: ' + str(now['contains'])) if now.get('contains') else ''} | | **looked at before touching** | {gaze_summary(it)} | | **lifecycle** | first {hms(lc.get('first_observed_sec'))} → last {hms(lc.get('last_observed_sec'))}, {lc.get('final_status')} | | **confidence** | {it.get('confidence')} | ### Where it is (location history) {fmt_list([state_sentence(x) for x in it.get('states') or []] or a.get('history'))} ### Moves (changes of place) {fmt_list([move_sentence(x) for x in it.get('transitions') or []])} ### Claimed look at the object before touching it (yellow lane; the red dot itself is always shown) {fmt_iv(ev.get('dwell')) if a.get('looked_at') else '- *(none claimed)*'} ### Narrations linked to this object (blue lane) — annotator text, NOT audio {fmt_iv(ev.get('narration')) if ev.get('narration') else fmt_list(a.get('said'))} ### Speech (purple lane) — what is actually spoken in the audio (ASR) {fmt_iv(ev.get('speech')) if ev.get('speech') else '- *(none: ' + str(it.get('speech_layer_note') or 'no speech detected') + ')*'} ### Sounds linked to this object (green lane) {fmt_iv(ev.get('sound')) if ev.get('sound') else fmt_list(a.get('heard'))} """ return md STATUS_ICON = {"match": "✅", "mismatch": "❌", "missing": "❌", "warn": "⚠️", "info": "ℹ️", "n/a": "➖"} def crosscheck_md(it): rows = it.get("crosscheck") or [] n_ok = sum(r["status"] == "match" for r in rows) n_bad = sum(r["status"] in ("mismatch", "missing") for r in rows) n_warn = sum(r["status"] == "warn" for r in rows) md = [f"## Automatic cross-check vs original HD-EPIC annotation — {n_ok} match, {n_bad} mismatch, {n_warn} warn", "| | what | memory (generated) | original HD-EPIC | note |", "|---|---|---|---|---|"] for r in rows: md.append(f"| {STATUS_ICON.get(r['status'], '')} | {r['what']} | {str(r['memory']).replace('|', '/')} | " f"{str(r['original']).replace('|', '/')} | {r.get('note', '')} |") md.append("\n### All HD-EPIC narrations in this clip window (✔ = attached to this object)") for n in it.get("raw_narrations_in_window") or []: md.append(f"- {'✔' if n['attached'] else ' '} **{hms(n['t0'])}–{hms(n['t1'])}** {n['text']}") return "\n".join(md) return md def clock_banner(it): """Big on-page clock above the video; a script keeps it in sync with playback.""" t0 = it["window_sec"][0] return (f'
') CLOCK_JS = """ () => { const fmt = (t) => { t = Math.round(t * 10) / 10; const h = Math.floor(t / 3600), m = Math.floor((t % 3600) / 60), s = t - h * 3600 - m * 60; return h + ":" + String(m).padStart(2, "0") + ":" + (s < 10 ? "0" : "") + s.toFixed(1); }; setInterval(() => { const v = document.querySelector("#eval-video video"), c = document.getElementById("live-clock"); if (v && c) c.textContent = fmt(parseFloat(c.dataset.t0) + (v.currentTime || 0)); }, 100); } """ def gallery_for(it): """Keyframe panels as plain