#!/usr/bin/env python """Gradio site for human evaluation of object-centric annotations. Run: python app.py [--port 7860] [--share] Responses are appended to responses/responses.csv. """ import argparse import csv import json import os import re from datetime import datetime from pathlib import Path import gradio as gr from filelock import FileLock HERE = Path(__file__).resolve().parent os.environ.setdefault("GRADIO_TEMP_DIR", str(HERE / ".gradio_cache")) MEDIA = HERE / "media" RESP_DIR = HERE / "responses" RESP_CSV = RESP_DIR / "responses.csv" LOCK = HERE / ".responses.csv.lock" CHOICES = ["Correct", "Partially correct", "Incorrect", "Can't tell"] # (column name, question shown to the annotator) DIMENSIONS = [ ("identity", "1. Identity — does `category` + `looks like` describe the object inside the red box?"), ("localization", "2. Localization — do the red box / green mask cover the right object (not a neighbour, hand, or background)?"), ("state_history", "3. Location & state history — is the where / relation / timing (`now`, `history`) correct?"), ("moves", "4. Moves — are the recorded movement events (from → to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"), ("sound_speech", "5. Sound & speech — do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"), ("gaze", "6. Gaze — is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"), ] # ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ---- DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses SCHEDULER = None if DATASET_REPO and os.environ.get("HF_TOKEN"): from huggingface_hub import CommitScheduler, HfApi, hf_hub_download RESP_DIR.mkdir(exist_ok=True) try: # resume from the copy in the dataset api = HfApi(token=os.environ["HF_TOKEN"]) api.create_repo(DATASET_REPO, repo_type="dataset", private=True, exist_ok=True) except Exception as e: print("could not reach the responses dataset:", type(e).__name__, str(e)[:120]) for _v in ("A", "B", "C"): # one file per questionnaire version try: cached = hf_hub_download(DATASET_REPO, f"responses_{_v}.csv", repo_type="dataset", token=os.environ["HF_TOKEN"]) (RESP_DIR / f"responses_{_v}.csv").write_bytes(Path(cached).read_bytes()) print(f"restored version {_v}: {sum(1 for _ in open(cached)) - 1} responses") except Exception as e: # first run: nothing to restore print(f"version {_v}: nothing restored ({type(e).__name__})") SCHEDULER = CommitScheduler(repo_id=DATASET_REPO, repo_type="dataset", folder_path=RESP_DIR, path_in_repo=".", every=2, private=True, token=os.environ["HF_TOKEN"], allow_patterns=["*.csv"]) ITEMS = json.load(open(MEDIA / "index.json")) # Narrations are judged one by one. An object has between 0 and NARR_MAX of them, so the # page holds NARR_MAX radio groups and shows only as many as the current object needs. NARR_MAX = max(1, max(len(it["events"]["narration"]) for it in ITEMS)) COLUMNS = (["timestamp", "annotator", "version", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS] + ["narration_ids"] + [f"narration_{k + 1}" for k in range(NARR_MAX)] + ["comment"]) NO_NARR_Q = ("No narration is linked to this object. Is that right — is there really no narration in this clip " "about it? (All narrations of the clip are listed in the cross-check panel.)") # ---- three shorter questionnaires -------------------------------------------------------- # 95 objects is too long for one sitting, so the clips are dealt into versions A / B / C # (whole clips, so an annotator judges every object of the clips they watch; balanced by # object count and spread over the videos). Each version writes its own CSV. VERSIONS = ["A", "B", "C"] def _split_versions(): clips = {} for i, it in enumerate(ITEMS): clips.setdefault((it["video_id"], it["chunk"]), []).append(i) load = {v: 0 for v in VERSIONS} out = {v: [] for v in VERSIONS} by_video = {} for key in clips: by_video.setdefault(key[0], []).append(key) for vid in sorted(by_video): # within a video, biggest clip first -> lightest version taken = set() for key in sorted(by_video[vid], key=lambda k: (-len(clips[k]), k)): cand = [v for v in VERSIONS if v not in taken] or VERSIONS v = min(cand, key=lambda x: (load[x], x)) taken.add(v) out[v] += clips[key] load[v] += len(clips[key]) return {v: sorted(ix) for v, ix in out.items()} VERSION_ITEMS = _split_versions() def resp_csv(version): return RESP_DIR / f"responses_{version}.csv" # A responses file written with a different column set (older form) is kept under another # name instead of being appended to with misaligned columns. for _p in [RESP_CSV] + [resp_csv(v) for v in VERSIONS]: if _p.exists(): with open(_p, newline="") as _f: _hdr = next(csv.reader(_f), None) if _hdr != COLUMNS or _p == RESP_CSV: _p.rename(RESP_DIR / f"{_p.stem}_old_{datetime.now():%Y%m%d_%H%M%S}.csv") def hms(t, dec=1): """Seconds of the full recording -> H:MM:SS.s (the unit shown everywhere on the site).""" if t is None: return "?" t = round(float(t), dec) h, m = int(t // 3600), int(t % 3600 // 60) s = t - h * 3600 - m * 60 return f"{h}:{m:02d}:{s:0{3 + dec if dec else 2}.{dec}f}" def place(anchor): """`P02_counter.002` -> `the counter #2`; `person_01` -> `the wearer's hand`.""" if not anchor or anchor == "unknown": return "an unknown place" if anchor.startswith("person"): return "the wearer's hand" kind, _, num = re.sub(r"^P\d\d_", "", anchor).partition(".") return f"the {kind.replace('_', ' ')}" + (f" #{int(num)}" if num.isdigit() else "") def where(rel, anchor): if rel == "held_by": return "held in the wearer's hand" return {"inside": "inside", "on": "on", "hanging_on": "hanging on"}.get(rel, "at") + " " + place(anchor) def state_sentence(st): a, b = st["interval_sec"] w = where(st["relation"], st["anchor"]) if b is None: return f"from {hms(a)} until the end of the clip: {w}" if abs(b - a) < 0.05: return f"at {hms(a)}: {w}" return f"from {hms(a)} to {hms(b)}: {w}" def move_sentence(tr): a, b = tr["interval_sec"] when = f"at {hms(a)}" if abs(b - a) < 0.05 else f"between {hms(a)} and {hms(b)}" return (f"{when} it goes from “{where(tr['from']['relation'], tr['from']['anchor'])}” to " f"“{where(tr['to']['relation'], tr['to']['anchor'])}” (action: {tr['event'].replace('_', ' ')})") def gaze_summary(it): a, att = it["annotation"], it.get("attention") or {} if a.get("looked_at"): gap = att.get("prime_gap_sec") return (f"Yes — first look at {hms(att.get('first_gaze_sec'))}" + (f", about {gap:.1f}s before the hand touches it" if isinstance(gap, (int, float)) else "")) return "No — the memory found no look at this object before the hand touches it" def _join(xs): return "; ".join(f"({i + 1}) {x}" for i, x in enumerate(xs)) def _bul(xs): return "\n".join(f"- {x}" for x in xs) def state_md(st): a, b = st["interval_sec"] w = f"**{where(st['relation'], st['anchor'])}**" if b is None: return f"{hms(a)} → end of clip: {w}" return f"at {hms(a)}: {w}" if abs(b - a) < 0.05 else f"{hms(a)} → {hms(b)}: {w}" def move_md(tr): a, b = tr["interval_sec"] when = f"at **{hms(a)}**" if abs(b - a) < 0.05 else f"**{hms(a)} → {hms(b)}**" return (f"{when}: {where(tr['from']['relation'], tr['from']['anchor'])} **→** " f"{where(tr['to']['relation'], tr['to']['anchor'])} ({tr['event'].replace('_', ' ')})") def dim_questions(it): """Markdown text of the six fixed questions for THIS object: short, claim in bold.""" a, ev, att = it["annotation"], it.get("events") or {}, it.get("attention") or {} obj = f"**{a.get('category')}**" q = {} q["identity"] = f"**1. Identity** — is the boxed object a {obj} that looks like “**{a.get('looks_like')}**”?" q["localization"] = f"**2. Box / mask** — does the **green mask** follow the {obj} **through the video**, and do the **red box** and mask cover it in the **keyframes (zoom on the right)**? (not a neighbour, the hand or background)" states = it.get("states") or [] q["state_history"] = (f"**3. Location** — is the {obj} where the memory says, at these times?\n" + _bul([state_md(x) for x in states] or (a.get("history") or []))) moves = it.get("transitions") or [] q["moves"] = ((f"**4. Moves** *(move = picked up / put down / put into / taken out)* — are these **right**, " f"and is **none missing**?\n" + _bul([move_md(x) for x in moves])) if moves else f"**4. Moves** *(move = picked up / put down / put into / taken out)* — the memory records " f"**no move**. Does the {obj} really **stay in place** for the whole clip?") snd = [f"**{e['label']}** at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("sound") or []] sp = [f"“{e['label']}” at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("speech") or []] q["sound_speech"] = ((f"**5. Sounds 🔊 (turn sound on)** — can you **hear** each one at that time, and is it made " f"**by / with the** {obj}?\n" + _bul(snd)) if snd else f"**5. Sounds 🔊 (turn sound on)** — the memory links **no sound**. Does the {obj} really " "make **no sound** in the clip?") if sp: q["sound_speech"] += "\n\nSpeech in the clip — transcribed correctly?\n" + _bul(sp) HEAD = ("**6. Gaze right before touching** — only the **moment right before the hand touches the object** " "counts, **not the whole clip** (the red dot is on screen all the time).\n\n") fg, gap = att.get("first_gaze_sec"), att.get("prime_gap_sec") t0, t1 = it["window_sec"] if not a.get("looked_at"): q["gaze"] = (HEAD + f"Memory: the wearer did **NOT look** at the {obj} before touching it. " "In the **2–3 s before the hand touches it**, does the red dot **stay off** the object?") elif not (t0 <= (fg or -1) <= t1): q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **{hms(fg)}**, which is **outside this clip** " f"({hms(t0, 0)}–{hms(t1, 0)}) → please answer **Can't tell**.") else: q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **≈ {hms(fg)}**" + (f" (**{gap:.1f}s before touching it**)" if isinstance(gap, (int, float)) else "") + ". **At that moment**, is the **red dot on (or right next to) the object**?") return [q[k] for k, _ in DIMENSIONS] def narrations_of(it): return sorted(it["events"]["narration"], key=lambda e: e["t0"]) def narration_updates(it): """For each of the NARR_MAX slots: (markdown update, radio update) for this object.""" ns = narrations_of(it) mds, rds = [], [] for k in range(NARR_MAX): if k < len(ns): e = ns[k] mds.append(gr.update(visible=True, value=f"**Narration {k + 1} / {len(ns)}** · **{hms(e['t0'])}–{hms(e['t1'])}** · " f"“{e['label']}” — about **this object**, at the **right time**?")) rds.append(gr.update(visible=True, value=None)) elif k == 0: mds.append(gr.update(visible=True, value="**No narration is linked** to this object. Is that right — does **no " "narration in this clip** talk about it? (all narrations: cross-check panel)")) rds.append(gr.update(visible=True, value=None)) else: mds.append(gr.update(visible=False, value="")) rds.append(gr.update(visible=False, value=None)) return mds + rds N = len(ITEMS) def item_key(it): return (it["video_id"], it["chunk"], it["object_id"]) # ---------------------------------------------------------------- csv i/o def load_done(annotator, version): done = set() p = resp_csv(version) if p.exists(): with open(p, newline="") as f: for r in csv.DictReader(f): if r["annotator"] == annotator: done.add((r["video_id"], r["chunk"], r["object_id"])) return done def _write(row, version): p = resp_csv(version) new = not p.exists() with open(p, "a", newline="") as f: w = csv.DictWriter(f, fieldnames=COLUMNS) if new: w.writeheader() w.writerow(row) def append_row(row, version): RESP_DIR.mkdir(exist_ok=True) with FileLock(str(LOCK)): if SCHEDULER is not None: with SCHEDULER.lock: _write(row, version) else: _write(row, version) if SCHEDULER is not None: # Push right away instead of waiting for the 2-minute tick: a restart of the Space # (every redeploy causes one) would otherwise lose whatever was not yet synced. try: SCHEDULER.trigger() except Exception as e: print("immediate sync failed:", type(e).__name__, str(e)[:120]) # ---------------------------------------------------------------- rendering def fmt_list(xs): return "\n".join(f"- {x}" for x in xs) if xs else "- *(none)*" def fmt_iv(evs): return "\n".join(f"- **{hms(e['t0'])}–{hms(e['t1'])}** {e['label']}" for e in evs) if evs else "- *(none)*" def annotation_md(it): a = it["annotation"] now = a.get("now") or {} lc = it.get("lifecycle") or {} ev = it.get("events") or {} att = it.get("attention") or {} md = f"""## `{it['object_id']}` **video** `{it['video_id']}` / {it['chunk']}   window {hms(it['window_sec'][0])}–{hms(it['window_sec'][1])} | field | value | |---|---| | **category** | {a.get('category')} | | **looks like** | {a.get('looks_like')} | | **color / material** | {a.get('color', '?')} / {a.get('material', '?')} | | **at the end of the clip** | {where(now.get('relation'), now.get('at'))} (since {hms(now.get('since_sec'))}){(' · contains: ' + str(now['contains'])) if now.get('contains') else ''} | | **looked at before touching** | {gaze_summary(it)} | | **lifecycle** | first {hms(lc.get('first_observed_sec'))} → last {hms(lc.get('last_observed_sec'))}, {lc.get('final_status')} | | **confidence** | {it.get('confidence')} | ### Where it is (location history) {fmt_list([state_sentence(x) for x in it.get('states') or []] or a.get('history'))} ### Moves (changes of place) {fmt_list([move_sentence(x) for x in it.get('transitions') or []])} ### Claimed look at the object before touching it (yellow lane; the red dot itself is always shown) {fmt_iv(ev.get('dwell')) if a.get('looked_at') else '- *(none claimed)*'} ### Narrations linked to this object (blue lane) — annotator text, NOT audio {fmt_iv(ev.get('narration')) if ev.get('narration') else fmt_list(a.get('said'))} ### Speech (purple lane) — what is actually spoken in the audio (ASR) {fmt_iv(ev.get('speech')) if ev.get('speech') else '- *(none: ' + str(it.get('speech_layer_note') or 'no speech detected') + ')*'} ### Sounds linked to this object (green lane) {fmt_iv(ev.get('sound')) if ev.get('sound') else fmt_list(a.get('heard'))} """ return md STATUS_ICON = {"match": "✅", "mismatch": "❌", "missing": "❌", "warn": "⚠️", "info": "ℹ️", "n/a": "➖"} def crosscheck_md(it): rows = it.get("crosscheck") or [] n_ok = sum(r["status"] == "match" for r in rows) n_bad = sum(r["status"] in ("mismatch", "missing") for r in rows) n_warn = sum(r["status"] == "warn" for r in rows) md = [f"## Automatic cross-check vs original HD-EPIC annotation — {n_ok} match, {n_bad} mismatch, {n_warn} warn", "| | what | memory (generated) | original HD-EPIC | note |", "|---|---|---|---|---|"] for r in rows: md.append(f"| {STATUS_ICON.get(r['status'], '')} | {r['what']} | {str(r['memory']).replace('|', '/')} | " f"{str(r['original']).replace('|', '/')} | {r.get('note', '')} |") md.append("\n### All HD-EPIC narrations in this clip window (✔ = attached to this object)") for n in it.get("raw_narrations_in_window") or []: md.append(f"- {'✔' if n['attached'] else '  '} **{hms(n['t0'])}–{hms(n['t1'])}** {n['text']}") return "\n".join(md) return md def clock_banner(it): """Big on-page clock above the video; a script keeps it in sync with playback.""" t0 = it["window_sec"][0] return (f'
⏱ VIDEO TIME' f'{hms(t0)}
' f'
Every time in the questions refers to this clock ' f'(hours : minutes : seconds of the recording; also in yellow at the top-right of the video).
' f'This clip runs {hms(t0, 0)} – {hms(it["window_sec"][1], 0)}.
') CLOCK_JS = """ () => { const fmt = (t) => { t = Math.round(t * 10) / 10; const h = Math.floor(t / 3600), m = Math.floor((t % 3600) / 60), s = t - h * 3600 - m * 60; return h + ":" + String(m).padStart(2, "0") + ":" + (s < 10 ? "0" : "") + s.toFixed(1); }; setInterval(() => { const v = document.querySelector("#eval-video video"), c = document.getElementById("live-clock"); if (v && c) c.textContent = fmt(parseFloat(c.dataset.t0) + (v.currentTime || 0)); }, 100); } """ def gallery_for(it): """Keyframe panels as plain tags (full width, stacked) -- the Gallery component crops wide images. Files are served by Gradio from the allowed media directory.""" out = [] for k in it["keyframes"]: url = f"/gradio_api/file={MEDIA / k['path']}" out.append(f'keyframe at {hms(k[') return "".join(out) or "no keyframe" def show(idx, annotator, version): """idx is the position inside the chosen version's item list.""" ix = VERSION_ITEMS[version] idx = max(0, min(idx, len(ix) - 1)) it = ITEMS[ix[idx]] done = load_done(annotator, version) if annotator else set() status = "✅ answered" if item_key(it) in done else "⬜ not answered" header = (f"### {annotator or '?'} · version {version} — item {idx + 1} / {len(ix)}   {status}   " f"({len(done)} / {len(ix)} done)") return ([idx, header, clock_banner(it), str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)] + dim_questions(it) + [None] * len(DIMENSIONS) + narration_updates(it) + [""]) def first_unanswered(annotator, version): done = load_done(annotator, version) for i, k in enumerate(VERSION_ITEMS[version]): if item_key(ITEMS[k]) not in done: return i return 0 # ---------------------------------------------------------------- app # Side-by-side layout: the video stays put on the left while the questions scroll on the # right, so nobody has to scroll away from the clip to answer. Stacks again on narrow screens. CSS = """ #eval-row { align-items: flex-start; flex-wrap: nowrap; } #left-pane, #right-pane { max-height: calc(100vh - 70px); overflow-y: auto; flex-wrap: nowrap; } /* a column is a flex box: without this its children are squeezed to fit instead of scrolling */ #left-pane > *, #right-pane > * { flex-shrink: 0; } #right-pane { padding-right: 10px; } #eval-video video { max-height: 56vh; width: 100%; object-fit: contain; } #kf-gallery img { cursor: zoom-in; } #vt-banner { background: #ffe14d; color: #111; border: 2px solid #111; border-radius: 8px; padding: 8px 12px; display: flex; align-items: center; gap: 14px; flex-wrap: wrap; } #vt-banner .vt-clock { display: flex; align-items: center; gap: 8px; font-size: 15px; white-space: nowrap; } #vt-banner #live-clock { font: 700 30px/1 ui-monospace, Menlo, Consolas, monospace; background: #111; color: #ffe14d; padding: 5px 10px; border-radius: 6px; } #vt-banner .vt-note { font-size: 13px; line-height: 1.35; flex: 1 1 240px; color: #111; } #vt-banner b { color: #111; } .q-stem { margin-top: 14px !important; margin-bottom: 2px !important; } .q-stem p, .q-stem li { font-size: 15px; line-height: 1.35; } .q-ans { margin-bottom: 6px !important; } @media (max-width: 1000px) { #eval-row { flex-wrap: wrap; } #left-pane, #right-pane { max-height: none; overflow-y: visible; } } """ with gr.Blocks(title="Object-centric annotation eval") as demo: idx_state = gr.State(0) name_state = gr.State("") ver_state = gr.State("A") with gr.Column(visible=True) as login_col: gr.Markdown("# Object-centric annotation evaluation\nEnter your name to start (progress is saved per name).") name_in = gr.Textbox(label="Your name", placeholder="e.g. alice") ver_in = gr.Radio(VERSIONS, value="A", label="Questionnaire version (the one you were assigned)", info=" · ".join(f"{v}: {len(VERSION_ITEMS[v])} objects" for v in VERSIONS)) start_btn = gr.Button("Start", variant="primary") with gr.Column(visible=False) as main_col: header = gr.Markdown() with gr.Row(elem_id="eval-row"): # ---- left: what to look at (stays in view) ---- with gr.Column(scale=5, elem_id="left-pane"): clock_html = gr.HTML() video = gr.Video(show_label=False, autoplay=True, loop=True, elem_id="eval-video") gr.Markdown("**Keyframes** — left: full frame · right: **zoom on the object** " "(green = mask, red = box, ring = gaze). Click an image to open it full size.") gallery = gr.HTML(elem_id="kf-gallery") with gr.Accordion("Legend — what the overlays mean", open=False): gr.Markdown( "- **VIDEO TIME** (top-right, boxed) = seconds of the full recording; every annotation time uses it " "(the player's own 0–30 s counter does not).\n" "- **green overlay** = the object's mask, tracked through the video (absent while the object is out of view) · " "**red box** = the annotated box, near each keyframe · **red dot** = where the wearer is looking.\n" "- yellow **GAZE ON OBJECT** = the memory claims a look at the object.\n" "- **blue NARRATION** = annotator text (not audio), with [start–end] · **purple SPEECH** = words " "actually spoken · **green SOUND** = sound events.\n" "- bottom bar = timeline: lanes sound / narration / speech / dwell, white ticks = narration " "start/end, red ticks = keyframes, white line = now.") # ---- right: the questions (scrolls on its own) ---- with gr.Column(scale=6, elem_id="right-pane"): with gr.Accordion("How to judge — read this once", open=True): gr.Markdown( "- Each question states **one claim** of an auto-generated memory about **one object**. Say whether it matches the video.\n" "- **Correct** / **Partially correct** (right idea, a place / time / detail is off) / **Incorrect** / **Can't tell**.\n" "- **Times** are **hours:minutes:seconds of the recording** → read the big yellow **VIDEO TIME** clock above the video. Differences under ~1 s are fine.\n" "- **Places** like “the counter #2”: judge the **kind of place**; ignore the number.\n" "- **Sound**: 🔊 turn it on (audio is amplified).\n" "- **Gaze**: the red dot is always shown; the gaze question is only about the **moment right before the hand touches the object**.") q_mds, radios = [], [] for key, _ in DIMENSIONS: q_mds.append(gr.Markdown(elem_classes="q-stem")) radios.append(gr.Radio(CHOICES, show_label=False, container=False, elem_classes="q-ans")) gr.Markdown("### Narrations — judge **each one** separately (blue subtitles; compare with **VIDEO TIME**)") narr_mds, narr_radios = [], [] for k in range(NARR_MAX): narr_mds.append(gr.Markdown(visible=(k == 0), elem_classes="q-stem")) narr_radios.append(gr.Radio(CHOICES, show_label=False, container=False, visible=(k == 0), elem_classes="q-ans")) comment = gr.Textbox(label="Comment (optional)", lines=2) with gr.Row(): prev_btn = gr.Button("◀ Prev") skip_btn = gr.Button("Skip ▶") submit_btn = gr.Button("Submit & Next ▶", variant="primary") msg = gr.Markdown() with gr.Accordion("Full memory record of this object (reference)", open=False): ann_md = gr.Markdown() with gr.Accordion("Cross-check vs original HD-EPIC annotation (reference)", open=False): xc_md = gr.Markdown() outputs = [idx_state, header, clock_html, video, gallery, ann_md, xc_md] + q_mds + radios + narr_mds + narr_radios + [comment] def start(name, version): name = (name or "").strip() if not name: raise gr.Error("Please enter your name.") if version not in VERSIONS: raise gr.Error("Please choose a questionnaire version.") i = first_unanswered(name, version) return [name, version, gr.update(visible=False), gr.update(visible=True)] + show(i, name, version) start_btn.click(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs) name_in.submit(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs) def preselect(request: gr.Request): """A link like .../?v=B pre-selects that version.""" v = (request.query_params.get("v") or "").upper() if request else "" return gr.update(value=v) if v in VERSIONS else gr.update() demo.load(preselect, None, [ver_in]) demo.load(None, None, None, js=CLOCK_JS) prev_btn.click(lambda i, n, v: show(i - 1, n, v), [idx_state, name_state, ver_state], outputs) skip_btn.click(lambda i, n, v: show(i + 1, n, v), [idx_state, name_state, ver_state], outputs) def submit(i, name, version, *vals): answers, narr_ans, cmt = vals[:len(DIMENSIONS)], vals[len(DIMENSIONS):-1], vals[-1] ix = VERSION_ITEMS[version] it = ITEMS[ix[i]] ns = narrations_of(it) need = max(1, len(ns)) if any(a is None for a in answers) or any(a is None for a in narr_ans[:need]): raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions and all {need} narration question(s) before submitting.") row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name, "version": version, "video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"], "comment": (cmt or "").strip()} row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)}) row["narration_ids"] = ";".join(str(e.get("id")) for e in ns) or "(none)" row.update({f"narration_{k + 1}": (narr_ans[k] if k < need else "") for k in range(NARR_MAX)}) append_row(row, version) if i + 1 >= len(ix): return show(i, name, version) + [f"**Saved.** That was the last item of version {version} — all {len(ix)} done. Thank you!"] return show(i + 1, name, version) + [f"Saved `{it['object_id']}` → {resp_csv(version).name}"] submit_btn.click(submit, [idx_state, name_state, ver_state] + radios + narr_radios + [comment], outputs + [msg]) if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--port", type=int, default=7860) ap.add_argument("--host", default="0.0.0.0") ap.add_argument("--share", action="store_true") a = ap.parse_args() demo.launch(server_name=a.host, server_port=a.port, share=a.share, allowed_paths=[str(MEDIA)], show_error=True, css=CSS)