Spaces:
Running
Running
Download app.py from WitneyWW/object-memory-eval: direct link, hf CLI and curl.
- Browser
- Download file 29.1 kB
-
https://huggingface.co/spaces/WitneyWW/object-memory-eval/resolve/main/app.py
- Command line
-
hf download hf://spaces/WitneyWW/object-memory-eval/app.py
-
curl -L -o app.py https://huggingface.co/spaces/WitneyWW/object-memory-eval/resolve/main/app.py
29.1 kB
| #!/usr/bin/env python | |
| """Gradio site for human evaluation of object-centric annotations. | |
| Run: python app.py [--port 7860] [--share] | |
| Responses are appended to responses/responses.csv. | |
| """ | |
| import argparse | |
| import csv | |
| import json | |
| import os | |
| import re | |
| from datetime import datetime | |
| from pathlib import Path | |
| import gradio as gr | |
| from filelock import FileLock | |
| HERE = Path(__file__).resolve().parent | |
| os.environ.setdefault("GRADIO_TEMP_DIR", str(HERE / ".gradio_cache")) | |
| MEDIA = HERE / "media" | |
| RESP_DIR = HERE / "responses" | |
| RESP_CSV = RESP_DIR / "responses.csv" | |
| LOCK = HERE / ".responses.csv.lock" | |
| CHOICES = ["Correct", "Partially correct", "Incorrect", "Can't tell"] | |
| # (column name, question shown to the annotator) | |
| DIMENSIONS = [ | |
| ("identity", "1. Identity — does `category` + `looks like` describe the object inside the red box?"), | |
| ("localization", "2. Localization — do the red box / green mask cover the right object (not a neighbour, hand, or background)?"), | |
| ("state_history", "3. Location & state history — is the where / relation / timing (`now`, `history`) correct?"), | |
| ("moves", "4. Moves — are the recorded movement events (from → to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"), | |
| ("sound_speech", "5. Sound & speech — do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"), | |
| ("gaze", "6. Gaze — is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"), | |
| ] | |
| # ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ---- | |
| DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses | |
| SCHEDULER = None | |
| if DATASET_REPO and os.environ.get("HF_TOKEN"): | |
| from huggingface_hub import CommitScheduler, HfApi, hf_hub_download | |
| RESP_DIR.mkdir(exist_ok=True) | |
| try: # resume from the copy in the dataset | |
| api = HfApi(token=os.environ["HF_TOKEN"]) | |
| api.create_repo(DATASET_REPO, repo_type="dataset", private=True, exist_ok=True) | |
| except Exception as e: | |
| print("could not reach the responses dataset:", type(e).__name__, str(e)[:120]) | |
| for _v in ("A", "B", "C"): # one file per questionnaire version | |
| try: | |
| cached = hf_hub_download(DATASET_REPO, f"responses_{_v}.csv", repo_type="dataset", | |
| token=os.environ["HF_TOKEN"]) | |
| (RESP_DIR / f"responses_{_v}.csv").write_bytes(Path(cached).read_bytes()) | |
| print(f"restored version {_v}: {sum(1 for _ in open(cached)) - 1} responses") | |
| except Exception as e: # first run: nothing to restore | |
| print(f"version {_v}: nothing restored ({type(e).__name__})") | |
| SCHEDULER = CommitScheduler(repo_id=DATASET_REPO, repo_type="dataset", folder_path=RESP_DIR, | |
| path_in_repo=".", every=2, private=True, token=os.environ["HF_TOKEN"], | |
| allow_patterns=["*.csv"]) | |
| ITEMS = json.load(open(MEDIA / "index.json")) | |
| # Narrations are judged one by one. An object has between 0 and NARR_MAX of them, so the | |
| # page holds NARR_MAX radio groups and shows only as many as the current object needs. | |
| NARR_MAX = max(1, max(len(it["events"]["narration"]) for it in ITEMS)) | |
| COLUMNS = (["timestamp", "annotator", "version", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS] | |
| + ["narration_ids"] + [f"narration_{k + 1}" for k in range(NARR_MAX)] + ["comment"]) | |
| NO_NARR_Q = ("No narration is linked to this object. Is that right — is there really no narration in this clip " | |
| "about it? (All narrations of the clip are listed in the cross-check panel.)") | |
| # ---- three shorter questionnaires -------------------------------------------------------- | |
| # 95 objects is too long for one sitting, so the clips are dealt into versions A / B / C | |
| # (whole clips, so an annotator judges every object of the clips they watch; balanced by | |
| # object count and spread over the videos). Each version writes its own CSV. | |
| VERSIONS = ["A", "B", "C"] | |
| def _split_versions(): | |
| clips = {} | |
| for i, it in enumerate(ITEMS): | |
| clips.setdefault((it["video_id"], it["chunk"]), []).append(i) | |
| load = {v: 0 for v in VERSIONS} | |
| out = {v: [] for v in VERSIONS} | |
| by_video = {} | |
| for key in clips: | |
| by_video.setdefault(key[0], []).append(key) | |
| for vid in sorted(by_video): # within a video, biggest clip first -> lightest version | |
| taken = set() | |
| for key in sorted(by_video[vid], key=lambda k: (-len(clips[k]), k)): | |
| cand = [v for v in VERSIONS if v not in taken] or VERSIONS | |
| v = min(cand, key=lambda x: (load[x], x)) | |
| taken.add(v) | |
| out[v] += clips[key] | |
| load[v] += len(clips[key]) | |
| return {v: sorted(ix) for v, ix in out.items()} | |
| VERSION_ITEMS = _split_versions() | |
| def resp_csv(version): | |
| return RESP_DIR / f"responses_{version}.csv" | |
| # A responses file written with a different column set (older form) is kept under another | |
| # name instead of being appended to with misaligned columns. | |
| for _p in [RESP_CSV] + [resp_csv(v) for v in VERSIONS]: | |
| if _p.exists(): | |
| with open(_p, newline="") as _f: | |
| _hdr = next(csv.reader(_f), None) | |
| if _hdr != COLUMNS or _p == RESP_CSV: | |
| _p.rename(RESP_DIR / f"{_p.stem}_old_{datetime.now():%Y%m%d_%H%M%S}.csv") | |
| def hms(t, dec=1): | |
| """Seconds of the full recording -> H:MM:SS.s (the unit shown everywhere on the site).""" | |
| if t is None: | |
| return "?" | |
| t = round(float(t), dec) | |
| h, m = int(t // 3600), int(t % 3600 // 60) | |
| s = t - h * 3600 - m * 60 | |
| return f"{h}:{m:02d}:{s:0{3 + dec if dec else 2}.{dec}f}" | |
| def place(anchor): | |
| """`P02_counter.002` -> `the counter #2`; `person_01` -> `the wearer's hand`.""" | |
| if not anchor or anchor == "unknown": | |
| return "an unknown place" | |
| if anchor.startswith("person"): | |
| return "the wearer's hand" | |
| kind, _, num = re.sub(r"^P\d\d_", "", anchor).partition(".") | |
| return f"the {kind.replace('_', ' ')}" + (f" #{int(num)}" if num.isdigit() else "") | |
| def where(rel, anchor): | |
| if rel == "held_by": | |
| return "held in the wearer's hand" | |
| return {"inside": "inside", "on": "on", "hanging_on": "hanging on"}.get(rel, "at") + " " + place(anchor) | |
| def state_sentence(st): | |
| a, b = st["interval_sec"] | |
| w = where(st["relation"], st["anchor"]) | |
| if b is None: | |
| return f"from {hms(a)} until the end of the clip: {w}" | |
| if abs(b - a) < 0.05: | |
| return f"at {hms(a)}: {w}" | |
| return f"from {hms(a)} to {hms(b)}: {w}" | |
| def move_sentence(tr): | |
| a, b = tr["interval_sec"] | |
| when = f"at {hms(a)}" if abs(b - a) < 0.05 else f"between {hms(a)} and {hms(b)}" | |
| return (f"{when} it goes from “{where(tr['from']['relation'], tr['from']['anchor'])}” to " | |
| f"“{where(tr['to']['relation'], tr['to']['anchor'])}” (action: {tr['event'].replace('_', ' ')})") | |
| def gaze_summary(it): | |
| a, att = it["annotation"], it.get("attention") or {} | |
| if a.get("looked_at"): | |
| gap = att.get("prime_gap_sec") | |
| return (f"Yes — first look at {hms(att.get('first_gaze_sec'))}" | |
| + (f", about {gap:.1f}s before the hand touches it" if isinstance(gap, (int, float)) else "")) | |
| return "No — the memory found no look at this object before the hand touches it" | |
| def _join(xs): | |
| return "; ".join(f"({i + 1}) {x}" for i, x in enumerate(xs)) | |
| def _bul(xs): | |
| return "\n".join(f"- {x}" for x in xs) | |
| def state_md(st): | |
| a, b = st["interval_sec"] | |
| w = f"**{where(st['relation'], st['anchor'])}**" | |
| if b is None: | |
| return f"{hms(a)} → end of clip: {w}" | |
| return f"at {hms(a)}: {w}" if abs(b - a) < 0.05 else f"{hms(a)} → {hms(b)}: {w}" | |
| def move_md(tr): | |
| a, b = tr["interval_sec"] | |
| when = f"at **{hms(a)}**" if abs(b - a) < 0.05 else f"**{hms(a)} → {hms(b)}**" | |
| return (f"{when}: {where(tr['from']['relation'], tr['from']['anchor'])} **→** " | |
| f"{where(tr['to']['relation'], tr['to']['anchor'])} ({tr['event'].replace('_', ' ')})") | |
| def dim_questions(it): | |
| """Markdown text of the six fixed questions for THIS object: short, claim in bold.""" | |
| a, ev, att = it["annotation"], it.get("events") or {}, it.get("attention") or {} | |
| obj = f"**{a.get('category')}**" | |
| q = {} | |
| q["identity"] = f"**1. Identity** — is the boxed object a {obj} that looks like “**{a.get('looks_like')}**”?" | |
| q["localization"] = f"**2. Box / mask** — does the **green mask** follow the {obj} **through the video**, and do the **red box** and mask cover it in the **keyframes (zoom on the right)**? (not a neighbour, the hand or background)" | |
| states = it.get("states") or [] | |
| q["state_history"] = (f"**3. Location** — is the {obj} where the memory says, at these times?\n" | |
| + _bul([state_md(x) for x in states] or (a.get("history") or []))) | |
| moves = it.get("transitions") or [] | |
| q["moves"] = ((f"**4. Moves** *(move = picked up / put down / put into / taken out)* — are these **right**, " | |
| f"and is **none missing**?\n" + _bul([move_md(x) for x in moves])) | |
| if moves else | |
| f"**4. Moves** *(move = picked up / put down / put into / taken out)* — the memory records " | |
| f"**no move**. Does the {obj} really **stay in place** for the whole clip?") | |
| snd = [f"**{e['label']}** at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("sound") or []] | |
| sp = [f"“{e['label']}” at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("speech") or []] | |
| q["sound_speech"] = ((f"**5. Sounds 🔊 (turn sound on)** — can you **hear** each one at that time, and is it made " | |
| f"**by / with the** {obj}?\n" + _bul(snd)) | |
| if snd else | |
| f"**5. Sounds 🔊 (turn sound on)** — the memory links **no sound**. Does the {obj} really " | |
| "make **no sound** in the clip?") | |
| if sp: | |
| q["sound_speech"] += "\n\nSpeech in the clip — transcribed correctly?\n" + _bul(sp) | |
| HEAD = ("**6. Gaze right before touching** — only the **moment right before the hand touches the object** " | |
| "counts, **not the whole clip** (the red dot is on screen all the time).\n\n") | |
| fg, gap = att.get("first_gaze_sec"), att.get("prime_gap_sec") | |
| t0, t1 = it["window_sec"] | |
| if not a.get("looked_at"): | |
| q["gaze"] = (HEAD + f"Memory: the wearer did **NOT look** at the {obj} before touching it. " | |
| "In the **2–3 s before the hand touches it**, does the red dot **stay off** the object?") | |
| elif not (t0 <= (fg or -1) <= t1): | |
| q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **{hms(fg)}**, which is **outside this clip** " | |
| f"({hms(t0, 0)}–{hms(t1, 0)}) → please answer **Can't tell**.") | |
| else: | |
| q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **≈ {hms(fg)}**" | |
| + (f" (**{gap:.1f}s before touching it**)" if isinstance(gap, (int, float)) else "") | |
| + ". **At that moment**, is the **red dot on (or right next to) the object**?") | |
| return [q[k] for k, _ in DIMENSIONS] | |
| def narrations_of(it): | |
| return sorted(it["events"]["narration"], key=lambda e: e["t0"]) | |
| def narration_updates(it): | |
| """For each of the NARR_MAX slots: (markdown update, radio update) for this object.""" | |
| ns = narrations_of(it) | |
| mds, rds = [], [] | |
| for k in range(NARR_MAX): | |
| if k < len(ns): | |
| e = ns[k] | |
| mds.append(gr.update(visible=True, value=f"**Narration {k + 1} / {len(ns)}** · **{hms(e['t0'])}–{hms(e['t1'])}** · " | |
| f"“{e['label']}” — about **this object**, at the **right time**?")) | |
| rds.append(gr.update(visible=True, value=None)) | |
| elif k == 0: | |
| mds.append(gr.update(visible=True, value="**No narration is linked** to this object. Is that right — does **no " | |
| "narration in this clip** talk about it? (all narrations: cross-check panel)")) | |
| rds.append(gr.update(visible=True, value=None)) | |
| else: | |
| mds.append(gr.update(visible=False, value="")) | |
| rds.append(gr.update(visible=False, value=None)) | |
| return mds + rds | |
| N = len(ITEMS) | |
| def item_key(it): | |
| return (it["video_id"], it["chunk"], it["object_id"]) | |
| # ---------------------------------------------------------------- csv i/o | |
| def load_done(annotator, version): | |
| done = set() | |
| p = resp_csv(version) | |
| if p.exists(): | |
| with open(p, newline="") as f: | |
| for r in csv.DictReader(f): | |
| if r["annotator"] == annotator: | |
| done.add((r["video_id"], r["chunk"], r["object_id"])) | |
| return done | |
| def _write(row, version): | |
| p = resp_csv(version) | |
| new = not p.exists() | |
| with open(p, "a", newline="") as f: | |
| w = csv.DictWriter(f, fieldnames=COLUMNS) | |
| if new: | |
| w.writeheader() | |
| w.writerow(row) | |
| def append_row(row, version): | |
| RESP_DIR.mkdir(exist_ok=True) | |
| with FileLock(str(LOCK)): | |
| if SCHEDULER is not None: | |
| with SCHEDULER.lock: | |
| _write(row, version) | |
| else: | |
| _write(row, version) | |
| if SCHEDULER is not None: | |
| # Push right away instead of waiting for the 2-minute tick: a restart of the Space | |
| # (every redeploy causes one) would otherwise lose whatever was not yet synced. | |
| try: | |
| SCHEDULER.trigger() | |
| except Exception as e: | |
| print("immediate sync failed:", type(e).__name__, str(e)[:120]) | |
| # ---------------------------------------------------------------- rendering | |
| def fmt_list(xs): | |
| return "\n".join(f"- {x}" for x in xs) if xs else "- *(none)*" | |
| def fmt_iv(evs): | |
| return "\n".join(f"- **{hms(e['t0'])}–{hms(e['t1'])}** {e['label']}" for e in evs) if evs else "- *(none)*" | |
| def annotation_md(it): | |
| a = it["annotation"] | |
| now = a.get("now") or {} | |
| lc = it.get("lifecycle") or {} | |
| ev = it.get("events") or {} | |
| att = it.get("attention") or {} | |
| md = f"""## `{it['object_id']}` | |
| **video** `{it['video_id']}` / {it['chunk']} window {hms(it['window_sec'][0])}–{hms(it['window_sec'][1])} | |
| | field | value | | |
| |---|---| | |
| | **category** | {a.get('category')} | | |
| | **looks like** | {a.get('looks_like')} | | |
| | **color / material** | {a.get('color', '?')} / {a.get('material', '?')} | | |
| | **at the end of the clip** | {where(now.get('relation'), now.get('at'))} (since {hms(now.get('since_sec'))}){(' · contains: ' + str(now['contains'])) if now.get('contains') else ''} | | |
| | **looked at before touching** | {gaze_summary(it)} | | |
| | **lifecycle** | first {hms(lc.get('first_observed_sec'))} → last {hms(lc.get('last_observed_sec'))}, {lc.get('final_status')} | | |
| | **confidence** | {it.get('confidence')} | | |
| ### Where it is (location history) | |
| {fmt_list([state_sentence(x) for x in it.get('states') or []] or a.get('history'))} | |
| ### Moves (changes of place) | |
| {fmt_list([move_sentence(x) for x in it.get('transitions') or []])} | |
| ### Claimed look at the object before touching it (yellow lane; the red dot itself is always shown) | |
| {fmt_iv(ev.get('dwell')) if a.get('looked_at') else '- *(none claimed)*'} | |
| ### Narrations linked to this object (blue lane) — annotator text, NOT audio | |
| {fmt_iv(ev.get('narration')) if ev.get('narration') else fmt_list(a.get('said'))} | |
| ### Speech (purple lane) — what is actually spoken in the audio (ASR) | |
| {fmt_iv(ev.get('speech')) if ev.get('speech') else '- *(none: ' + str(it.get('speech_layer_note') or 'no speech detected') + ')*'} | |
| ### Sounds linked to this object (green lane) | |
| {fmt_iv(ev.get('sound')) if ev.get('sound') else fmt_list(a.get('heard'))} | |
| """ | |
| return md | |
| STATUS_ICON = {"match": "✅", "mismatch": "❌", "missing": "❌", "warn": "⚠️", "info": "ℹ️", "n/a": "➖"} | |
| def crosscheck_md(it): | |
| rows = it.get("crosscheck") or [] | |
| n_ok = sum(r["status"] == "match" for r in rows) | |
| n_bad = sum(r["status"] in ("mismatch", "missing") for r in rows) | |
| n_warn = sum(r["status"] == "warn" for r in rows) | |
| md = [f"## Automatic cross-check vs original HD-EPIC annotation — {n_ok} match, {n_bad} mismatch, {n_warn} warn", | |
| "| | what | memory (generated) | original HD-EPIC | note |", "|---|---|---|---|---|"] | |
| for r in rows: | |
| md.append(f"| {STATUS_ICON.get(r['status'], '')} | {r['what']} | {str(r['memory']).replace('|', '/')} | " | |
| f"{str(r['original']).replace('|', '/')} | {r.get('note', '')} |") | |
| md.append("\n### All HD-EPIC narrations in this clip window (✔ = attached to this object)") | |
| for n in it.get("raw_narrations_in_window") or []: | |
| md.append(f"- {'✔' if n['attached'] else ' '} **{hms(n['t0'])}–{hms(n['t1'])}** {n['text']}") | |
| return "\n".join(md) | |
| return md | |
| def clock_banner(it): | |
| """Big on-page clock above the video; a script keeps it in sync with playback.""" | |
| t0 = it["window_sec"][0] | |
| return (f'<div id="vt-banner"><div class="vt-clock">⏱ <b>VIDEO TIME</b>' | |
| f'<span id="live-clock" data-t0="{t0}">{hms(t0)}</span></div>' | |
| f'<div class="vt-note"><b>Every time in the questions refers to this clock</b> ' | |
| f'(hours : minutes : seconds of the recording; also in yellow at the top-right of the video).<br>' | |
| f'This clip runs {hms(t0, 0)} – {hms(it["window_sec"][1], 0)}.</div></div>') | |
| CLOCK_JS = """ | |
| () => { | |
| const fmt = (t) => { | |
| t = Math.round(t * 10) / 10; | |
| const h = Math.floor(t / 3600), m = Math.floor((t % 3600) / 60), s = t - h * 3600 - m * 60; | |
| return h + ":" + String(m).padStart(2, "0") + ":" + (s < 10 ? "0" : "") + s.toFixed(1); | |
| }; | |
| setInterval(() => { | |
| const v = document.querySelector("#eval-video video"), c = document.getElementById("live-clock"); | |
| if (v && c) c.textContent = fmt(parseFloat(c.dataset.t0) + (v.currentTime || 0)); | |
| }, 100); | |
| } | |
| """ | |
| def gallery_for(it): | |
| """Keyframe panels as plain <img> tags (full width, stacked) -- the Gallery component | |
| crops wide images. Files are served by Gradio from the allowed media directory.""" | |
| out = [] | |
| for k in it["keyframes"]: | |
| url = f"/gradio_api/file={MEDIA / k['path']}" | |
| out.append(f'<a href="{url}" target="_blank"><img src="{url}" alt="keyframe at {hms(k["t"])}" ' | |
| f'style="width:100%;display:block;margin:0 0 8px 0;border-radius:6px"></a>') | |
| return "".join(out) or "<i>no keyframe</i>" | |
| def show(idx, annotator, version): | |
| """idx is the position inside the chosen version's item list.""" | |
| ix = VERSION_ITEMS[version] | |
| idx = max(0, min(idx, len(ix) - 1)) | |
| it = ITEMS[ix[idx]] | |
| done = load_done(annotator, version) if annotator else set() | |
| status = "✅ answered" if item_key(it) in done else "⬜ not answered" | |
| header = (f"### {annotator or '?'} · version {version} — item {idx + 1} / {len(ix)} {status} " | |
| f"({len(done)} / {len(ix)} done)") | |
| return ([idx, header, clock_banner(it), str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)] | |
| + dim_questions(it) + [None] * len(DIMENSIONS) + narration_updates(it) + [""]) | |
| def first_unanswered(annotator, version): | |
| done = load_done(annotator, version) | |
| for i, k in enumerate(VERSION_ITEMS[version]): | |
| if item_key(ITEMS[k]) not in done: | |
| return i | |
| return 0 | |
| # ---------------------------------------------------------------- app | |
| # Side-by-side layout: the video stays put on the left while the questions scroll on the | |
| # right, so nobody has to scroll away from the clip to answer. Stacks again on narrow screens. | |
| CSS = """ | |
| #eval-row { align-items: flex-start; flex-wrap: nowrap; } | |
| #left-pane, #right-pane { max-height: calc(100vh - 70px); overflow-y: auto; flex-wrap: nowrap; } | |
| /* a column is a flex box: without this its children are squeezed to fit instead of scrolling */ | |
| #left-pane > *, #right-pane > * { flex-shrink: 0; } | |
| #right-pane { padding-right: 10px; } | |
| #eval-video video { max-height: 56vh; width: 100%; object-fit: contain; } | |
| #kf-gallery img { cursor: zoom-in; } | |
| #vt-banner { background: #ffe14d; color: #111; border: 2px solid #111; border-radius: 8px; padding: 8px 12px; | |
| display: flex; align-items: center; gap: 14px; flex-wrap: wrap; } | |
| #vt-banner .vt-clock { display: flex; align-items: center; gap: 8px; font-size: 15px; white-space: nowrap; } | |
| #vt-banner #live-clock { font: 700 30px/1 ui-monospace, Menlo, Consolas, monospace; | |
| background: #111; color: #ffe14d; padding: 5px 10px; border-radius: 6px; } | |
| #vt-banner .vt-note { font-size: 13px; line-height: 1.35; flex: 1 1 240px; color: #111; } | |
| #vt-banner b { color: #111; } | |
| .q-stem { margin-top: 14px !important; margin-bottom: 2px !important; } | |
| .q-stem p, .q-stem li { font-size: 15px; line-height: 1.35; } | |
| .q-ans { margin-bottom: 6px !important; } | |
| @media (max-width: 1000px) { | |
| #eval-row { flex-wrap: wrap; } | |
| #left-pane, #right-pane { max-height: none; overflow-y: visible; } | |
| } | |
| """ | |
| with gr.Blocks(title="Object-centric annotation eval") as demo: | |
| idx_state = gr.State(0) | |
| name_state = gr.State("") | |
| ver_state = gr.State("A") | |
| with gr.Column(visible=True) as login_col: | |
| gr.Markdown("# Object-centric annotation evaluation\nEnter your name to start (progress is saved per name).") | |
| name_in = gr.Textbox(label="Your name", placeholder="e.g. alice") | |
| ver_in = gr.Radio(VERSIONS, value="A", label="Questionnaire version (the one you were assigned)", | |
| info=" · ".join(f"{v}: {len(VERSION_ITEMS[v])} objects" for v in VERSIONS)) | |
| start_btn = gr.Button("Start", variant="primary") | |
| with gr.Column(visible=False) as main_col: | |
| header = gr.Markdown() | |
| with gr.Row(elem_id="eval-row"): | |
| # ---- left: what to look at (stays in view) ---- | |
| with gr.Column(scale=5, elem_id="left-pane"): | |
| clock_html = gr.HTML() | |
| video = gr.Video(show_label=False, autoplay=True, loop=True, elem_id="eval-video") | |
| gr.Markdown("**Keyframes** — left: full frame · right: **zoom on the object** " | |
| "(green = mask, red = box, ring = gaze). Click an image to open it full size.") | |
| gallery = gr.HTML(elem_id="kf-gallery") | |
| with gr.Accordion("Legend — what the overlays mean", open=False): | |
| gr.Markdown( | |
| "- **VIDEO TIME** (top-right, boxed) = seconds of the full recording; every annotation time uses it " | |
| "(the player's own 0–30 s counter does not).\n" | |
| "- **green overlay** = the object's mask, tracked through the video (absent while the object is out of view) · " | |
| "**red box** = the annotated box, near each keyframe · **red dot** = where the wearer is looking.\n" | |
| "- yellow **GAZE ON OBJECT** = the memory claims a look at the object.\n" | |
| "- **blue NARRATION** = annotator text (not audio), with [start–end] · **purple SPEECH** = words " | |
| "actually spoken · **green SOUND** = sound events.\n" | |
| "- bottom bar = timeline: lanes sound / narration / speech / dwell, white ticks = narration " | |
| "start/end, red ticks = keyframes, white line = now.") | |
| # ---- right: the questions (scrolls on its own) ---- | |
| with gr.Column(scale=6, elem_id="right-pane"): | |
| with gr.Accordion("How to judge — read this once", open=True): | |
| gr.Markdown( | |
| "- Each question states **one claim** of an auto-generated memory about **one object**. Say whether it matches the video.\n" | |
| "- **Correct** / **Partially correct** (right idea, a place / time / detail is off) / **Incorrect** / **Can't tell**.\n" | |
| "- **Times** are **hours:minutes:seconds of the recording** → read the big yellow **VIDEO TIME** clock above the video. Differences under ~1 s are fine.\n" | |
| "- **Places** like “the counter #2”: judge the **kind of place**; ignore the number.\n" | |
| "- **Sound**: 🔊 turn it on (audio is amplified).\n" | |
| "- **Gaze**: the red dot is always shown; the gaze question is only about the **moment right before the hand touches the object**.") | |
| q_mds, radios = [], [] | |
| for key, _ in DIMENSIONS: | |
| q_mds.append(gr.Markdown(elem_classes="q-stem")) | |
| radios.append(gr.Radio(CHOICES, show_label=False, container=False, elem_classes="q-ans")) | |
| gr.Markdown("### Narrations — judge **each one** separately (blue subtitles; compare with **VIDEO TIME**)") | |
| narr_mds, narr_radios = [], [] | |
| for k in range(NARR_MAX): | |
| narr_mds.append(gr.Markdown(visible=(k == 0), elem_classes="q-stem")) | |
| narr_radios.append(gr.Radio(CHOICES, show_label=False, container=False, visible=(k == 0), | |
| elem_classes="q-ans")) | |
| comment = gr.Textbox(label="Comment (optional)", lines=2) | |
| with gr.Row(): | |
| prev_btn = gr.Button("◀ Prev") | |
| skip_btn = gr.Button("Skip ▶") | |
| submit_btn = gr.Button("Submit & Next ▶", variant="primary") | |
| msg = gr.Markdown() | |
| with gr.Accordion("Full memory record of this object (reference)", open=False): | |
| ann_md = gr.Markdown() | |
| with gr.Accordion("Cross-check vs original HD-EPIC annotation (reference)", open=False): | |
| xc_md = gr.Markdown() | |
| outputs = [idx_state, header, clock_html, video, gallery, ann_md, xc_md] + q_mds + radios + narr_mds + narr_radios + [comment] | |
| def start(name, version): | |
| name = (name or "").strip() | |
| if not name: | |
| raise gr.Error("Please enter your name.") | |
| if version not in VERSIONS: | |
| raise gr.Error("Please choose a questionnaire version.") | |
| i = first_unanswered(name, version) | |
| return [name, version, gr.update(visible=False), gr.update(visible=True)] + show(i, name, version) | |
| start_btn.click(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs) | |
| name_in.submit(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs) | |
| def preselect(request: gr.Request): | |
| """A link like .../?v=B pre-selects that version.""" | |
| v = (request.query_params.get("v") or "").upper() if request else "" | |
| return gr.update(value=v) if v in VERSIONS else gr.update() | |
| demo.load(preselect, None, [ver_in]) | |
| demo.load(None, None, None, js=CLOCK_JS) | |
| prev_btn.click(lambda i, n, v: show(i - 1, n, v), [idx_state, name_state, ver_state], outputs) | |
| skip_btn.click(lambda i, n, v: show(i + 1, n, v), [idx_state, name_state, ver_state], outputs) | |
| def submit(i, name, version, *vals): | |
| answers, narr_ans, cmt = vals[:len(DIMENSIONS)], vals[len(DIMENSIONS):-1], vals[-1] | |
| ix = VERSION_ITEMS[version] | |
| it = ITEMS[ix[i]] | |
| ns = narrations_of(it) | |
| need = max(1, len(ns)) | |
| if any(a is None for a in answers) or any(a is None for a in narr_ans[:need]): | |
| raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions and all {need} narration question(s) before submitting.") | |
| row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name, "version": version, | |
| "video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"], | |
| "comment": (cmt or "").strip()} | |
| row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)}) | |
| row["narration_ids"] = ";".join(str(e.get("id")) for e in ns) or "(none)" | |
| row.update({f"narration_{k + 1}": (narr_ans[k] if k < need else "") for k in range(NARR_MAX)}) | |
| append_row(row, version) | |
| if i + 1 >= len(ix): | |
| return show(i, name, version) + [f"**Saved.** That was the last item of version {version} — all {len(ix)} done. Thank you!"] | |
| return show(i + 1, name, version) + [f"Saved `{it['object_id']}` → {resp_csv(version).name}"] | |
| submit_btn.click(submit, [idx_state, name_state, ver_state] + radios + narr_radios + [comment], outputs + [msg]) | |
| if __name__ == "__main__": | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--port", type=int, default=7860) | |
| ap.add_argument("--host", default="0.0.0.0") | |
| ap.add_argument("--share", action="store_true") | |
| a = ap.parse_args() | |
| demo.launch(server_name=a.host, server_port=a.port, share=a.share, allowed_paths=[str(MEDIA)], show_error=True, css=CSS) | |