WitneyWW's picture
green mask tracked through the video (SAM 3 tracker); sync every submission immediately
a359c5e verified
Raw History Blame Contribute Delete
29.1 kB
#!/usr/bin/env python
"""Gradio site for human evaluation of object-centric annotations.
Run: python app.py [--port 7860] [--share]
Responses are appended to responses/responses.csv.
"""
import argparse
import csv
import json
import os
import re
from datetime import datetime
from pathlib import Path
import gradio as gr
from filelock import FileLock
HERE = Path(__file__).resolve().parent
os.environ.setdefault("GRADIO_TEMP_DIR", str(HERE / ".gradio_cache"))
MEDIA = HERE / "media"
RESP_DIR = HERE / "responses"
RESP_CSV = RESP_DIR / "responses.csv"
LOCK = HERE / ".responses.csv.lock"
CHOICES = ["Correct", "Partially correct", "Incorrect", "Can't tell"]
# (column name, question shown to the annotator)
DIMENSIONS = [
("identity", "1. Identity — does `category` + `looks like` describe the object inside the red box?"),
("localization", "2. Localization — do the red box / green mask cover the right object (not a neighbour, hand, or background)?"),
("state_history", "3. Location & state history — is the where / relation / timing (`now`, `history`) correct?"),
("moves", "4. Moves — are the recorded movement events (from → to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"),
("sound_speech", "5. Sound & speech — do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"),
("gaze", "6. Gaze — is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"),
]
# ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ----
DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses
SCHEDULER = None
if DATASET_REPO and os.environ.get("HF_TOKEN"):
from huggingface_hub import CommitScheduler, HfApi, hf_hub_download
RESP_DIR.mkdir(exist_ok=True)
try: # resume from the copy in the dataset
api = HfApi(token=os.environ["HF_TOKEN"])
api.create_repo(DATASET_REPO, repo_type="dataset", private=True, exist_ok=True)
except Exception as e:
print("could not reach the responses dataset:", type(e).__name__, str(e)[:120])
for _v in ("A", "B", "C"): # one file per questionnaire version
try:
cached = hf_hub_download(DATASET_REPO, f"responses_{_v}.csv", repo_type="dataset",
token=os.environ["HF_TOKEN"])
(RESP_DIR / f"responses_{_v}.csv").write_bytes(Path(cached).read_bytes())
print(f"restored version {_v}: {sum(1 for _ in open(cached)) - 1} responses")
except Exception as e: # first run: nothing to restore
print(f"version {_v}: nothing restored ({type(e).__name__})")
SCHEDULER = CommitScheduler(repo_id=DATASET_REPO, repo_type="dataset", folder_path=RESP_DIR,
path_in_repo=".", every=2, private=True, token=os.environ["HF_TOKEN"],
allow_patterns=["*.csv"])
ITEMS = json.load(open(MEDIA / "index.json"))
# Narrations are judged one by one. An object has between 0 and NARR_MAX of them, so the
# page holds NARR_MAX radio groups and shows only as many as the current object needs.
NARR_MAX = max(1, max(len(it["events"]["narration"]) for it in ITEMS))
COLUMNS = (["timestamp", "annotator", "version", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS]
+ ["narration_ids"] + [f"narration_{k + 1}" for k in range(NARR_MAX)] + ["comment"])
NO_NARR_Q = ("No narration is linked to this object. Is that right — is there really no narration in this clip "
"about it? (All narrations of the clip are listed in the cross-check panel.)")
# ---- three shorter questionnaires --------------------------------------------------------
# 95 objects is too long for one sitting, so the clips are dealt into versions A / B / C
# (whole clips, so an annotator judges every object of the clips they watch; balanced by
# object count and spread over the videos). Each version writes its own CSV.
VERSIONS = ["A", "B", "C"]
def _split_versions():
clips = {}
for i, it in enumerate(ITEMS):
clips.setdefault((it["video_id"], it["chunk"]), []).append(i)
load = {v: 0 for v in VERSIONS}
out = {v: [] for v in VERSIONS}
by_video = {}
for key in clips:
by_video.setdefault(key[0], []).append(key)
for vid in sorted(by_video): # within a video, biggest clip first -> lightest version
taken = set()
for key in sorted(by_video[vid], key=lambda k: (-len(clips[k]), k)):
cand = [v for v in VERSIONS if v not in taken] or VERSIONS
v = min(cand, key=lambda x: (load[x], x))
taken.add(v)
out[v] += clips[key]
load[v] += len(clips[key])
return {v: sorted(ix) for v, ix in out.items()}
VERSION_ITEMS = _split_versions()
def resp_csv(version):
return RESP_DIR / f"responses_{version}.csv"
# A responses file written with a different column set (older form) is kept under another
# name instead of being appended to with misaligned columns.
for _p in [RESP_CSV] + [resp_csv(v) for v in VERSIONS]:
if _p.exists():
with open(_p, newline="") as _f:
_hdr = next(csv.reader(_f), None)
if _hdr != COLUMNS or _p == RESP_CSV:
_p.rename(RESP_DIR / f"{_p.stem}_old_{datetime.now():%Y%m%d_%H%M%S}.csv")
def hms(t, dec=1):
"""Seconds of the full recording -> H:MM:SS.s (the unit shown everywhere on the site)."""
if t is None:
return "?"
t = round(float(t), dec)
h, m = int(t // 3600), int(t % 3600 // 60)
s = t - h * 3600 - m * 60
return f"{h}:{m:02d}:{s:0{3 + dec if dec else 2}.{dec}f}"
def place(anchor):
"""`P02_counter.002` -> `the counter #2`; `person_01` -> `the wearer's hand`."""
if not anchor or anchor == "unknown":
return "an unknown place"
if anchor.startswith("person"):
return "the wearer's hand"
kind, _, num = re.sub(r"^P\d\d_", "", anchor).partition(".")
return f"the {kind.replace('_', ' ')}" + (f" #{int(num)}" if num.isdigit() else "")
def where(rel, anchor):
if rel == "held_by":
return "held in the wearer's hand"
return {"inside": "inside", "on": "on", "hanging_on": "hanging on"}.get(rel, "at") + " " + place(anchor)
def state_sentence(st):
a, b = st["interval_sec"]
w = where(st["relation"], st["anchor"])
if b is None:
return f"from {hms(a)} until the end of the clip: {w}"
if abs(b - a) < 0.05:
return f"at {hms(a)}: {w}"
return f"from {hms(a)} to {hms(b)}: {w}"
def move_sentence(tr):
a, b = tr["interval_sec"]
when = f"at {hms(a)}" if abs(b - a) < 0.05 else f"between {hms(a)} and {hms(b)}"
return (f"{when} it goes from “{where(tr['from']['relation'], tr['from']['anchor'])}” to "
f"“{where(tr['to']['relation'], tr['to']['anchor'])}” (action: {tr['event'].replace('_', ' ')})")
def gaze_summary(it):
a, att = it["annotation"], it.get("attention") or {}
if a.get("looked_at"):
gap = att.get("prime_gap_sec")
return (f"Yes — first look at {hms(att.get('first_gaze_sec'))}"
+ (f", about {gap:.1f}s before the hand touches it" if isinstance(gap, (int, float)) else ""))
return "No — the memory found no look at this object before the hand touches it"
def _join(xs):
return "; ".join(f"({i + 1}) {x}" for i, x in enumerate(xs))
def _bul(xs):
return "\n".join(f"- {x}" for x in xs)
def state_md(st):
a, b = st["interval_sec"]
w = f"**{where(st['relation'], st['anchor'])}**"
if b is None:
return f"{hms(a)} → end of clip: {w}"
return f"at {hms(a)}: {w}" if abs(b - a) < 0.05 else f"{hms(a)} → {hms(b)}: {w}"
def move_md(tr):
a, b = tr["interval_sec"]
when = f"at **{hms(a)}**" if abs(b - a) < 0.05 else f"**{hms(a)} → {hms(b)}**"
return (f"{when}: {where(tr['from']['relation'], tr['from']['anchor'])} **→** "
f"{where(tr['to']['relation'], tr['to']['anchor'])} ({tr['event'].replace('_', ' ')})")
def dim_questions(it):
"""Markdown text of the six fixed questions for THIS object: short, claim in bold."""
a, ev, att = it["annotation"], it.get("events") or {}, it.get("attention") or {}
obj = f"**{a.get('category')}**"
q = {}
q["identity"] = f"**1. Identity** — is the boxed object a {obj} that looks like “**{a.get('looks_like')}**”?"
q["localization"] = f"**2. Box / mask** — does the **green mask** follow the {obj} **through the video**, and do the **red box** and mask cover it in the **keyframes (zoom on the right)**? (not a neighbour, the hand or background)"
states = it.get("states") or []
q["state_history"] = (f"**3. Location** — is the {obj} where the memory says, at these times?\n"
+ _bul([state_md(x) for x in states] or (a.get("history") or [])))
moves = it.get("transitions") or []
q["moves"] = ((f"**4. Moves** *(move = picked up / put down / put into / taken out)* — are these **right**, "
f"and is **none missing**?\n" + _bul([move_md(x) for x in moves]))
if moves else
f"**4. Moves** *(move = picked up / put down / put into / taken out)* — the memory records "
f"**no move**. Does the {obj} really **stay in place** for the whole clip?")
snd = [f"**{e['label']}** at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("sound") or []]
sp = [f"“{e['label']}” at {hms(e['t0'])}–{hms(e['t1'])}" for e in ev.get("speech") or []]
q["sound_speech"] = ((f"**5. Sounds 🔊 (turn sound on)** — can you **hear** each one at that time, and is it made "
f"**by / with the** {obj}?\n" + _bul(snd))
if snd else
f"**5. Sounds 🔊 (turn sound on)** — the memory links **no sound**. Does the {obj} really "
"make **no sound** in the clip?")
if sp:
q["sound_speech"] += "\n\nSpeech in the clip — transcribed correctly?\n" + _bul(sp)
HEAD = ("**6. Gaze right before touching** — only the **moment right before the hand touches the object** "
"counts, **not the whole clip** (the red dot is on screen all the time).\n\n")
fg, gap = att.get("first_gaze_sec"), att.get("prime_gap_sec")
t0, t1 = it["window_sec"]
if not a.get("looked_at"):
q["gaze"] = (HEAD + f"Memory: the wearer did **NOT look** at the {obj} before touching it. "
"In the **2–3 s before the hand touches it**, does the red dot **stay off** the object?")
elif not (t0 <= (fg or -1) <= t1):
q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **{hms(fg)}**, which is **outside this clip** "
f"({hms(t0, 0)}–{hms(t1, 0)}) → please answer **Can't tell**.")
else:
q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **≈ {hms(fg)}**"
+ (f" (**{gap:.1f}s before touching it**)" if isinstance(gap, (int, float)) else "")
+ ". **At that moment**, is the **red dot on (or right next to) the object**?")
return [q[k] for k, _ in DIMENSIONS]
def narrations_of(it):
return sorted(it["events"]["narration"], key=lambda e: e["t0"])
def narration_updates(it):
"""For each of the NARR_MAX slots: (markdown update, radio update) for this object."""
ns = narrations_of(it)
mds, rds = [], []
for k in range(NARR_MAX):
if k < len(ns):
e = ns[k]
mds.append(gr.update(visible=True, value=f"**Narration {k + 1} / {len(ns)}** · **{hms(e['t0'])}–{hms(e['t1'])}** · "
f"“{e['label']}” — about **this object**, at the **right time**?"))
rds.append(gr.update(visible=True, value=None))
elif k == 0:
mds.append(gr.update(visible=True, value="**No narration is linked** to this object. Is that right — does **no "
"narration in this clip** talk about it? (all narrations: cross-check panel)"))
rds.append(gr.update(visible=True, value=None))
else:
mds.append(gr.update(visible=False, value=""))
rds.append(gr.update(visible=False, value=None))
return mds + rds
N = len(ITEMS)
def item_key(it):
return (it["video_id"], it["chunk"], it["object_id"])
# ---------------------------------------------------------------- csv i/o
def load_done(annotator, version):
done = set()
p = resp_csv(version)
if p.exists():
with open(p, newline="") as f:
for r in csv.DictReader(f):
if r["annotator"] == annotator:
done.add((r["video_id"], r["chunk"], r["object_id"]))
return done
def _write(row, version):
p = resp_csv(version)
new = not p.exists()
with open(p, "a", newline="") as f:
w = csv.DictWriter(f, fieldnames=COLUMNS)
if new:
w.writeheader()
w.writerow(row)
def append_row(row, version):
RESP_DIR.mkdir(exist_ok=True)
with FileLock(str(LOCK)):
if SCHEDULER is not None:
with SCHEDULER.lock:
_write(row, version)
else:
_write(row, version)
if SCHEDULER is not None:
# Push right away instead of waiting for the 2-minute tick: a restart of the Space
# (every redeploy causes one) would otherwise lose whatever was not yet synced.
try:
SCHEDULER.trigger()
except Exception as e:
print("immediate sync failed:", type(e).__name__, str(e)[:120])
# ---------------------------------------------------------------- rendering
def fmt_list(xs):
return "\n".join(f"- {x}" for x in xs) if xs else "- *(none)*"
def fmt_iv(evs):
return "\n".join(f"- **{hms(e['t0'])}–{hms(e['t1'])}** {e['label']}" for e in evs) if evs else "- *(none)*"
def annotation_md(it):
a = it["annotation"]
now = a.get("now") or {}
lc = it.get("lifecycle") or {}
ev = it.get("events") or {}
att = it.get("attention") or {}
md = f"""## `{it['object_id']}`
**video** `{it['video_id']}` / {it['chunk']} &nbsp; window {hms(it['window_sec'][0])}–{hms(it['window_sec'][1])}
| field | value |
|---|---|
| **category** | {a.get('category')} |
| **looks like** | {a.get('looks_like')} |
| **color / material** | {a.get('color', '?')} / {a.get('material', '?')} |
| **at the end of the clip** | {where(now.get('relation'), now.get('at'))} (since {hms(now.get('since_sec'))}){(' · contains: ' + str(now['contains'])) if now.get('contains') else ''} |
| **looked at before touching** | {gaze_summary(it)} |
| **lifecycle** | first {hms(lc.get('first_observed_sec'))} → last {hms(lc.get('last_observed_sec'))}, {lc.get('final_status')} |
| **confidence** | {it.get('confidence')} |
### Where it is (location history)
{fmt_list([state_sentence(x) for x in it.get('states') or []] or a.get('history'))}
### Moves (changes of place)
{fmt_list([move_sentence(x) for x in it.get('transitions') or []])}
### Claimed look at the object before touching it (yellow lane; the red dot itself is always shown)
{fmt_iv(ev.get('dwell')) if a.get('looked_at') else '- *(none claimed)*'}
### Narrations linked to this object (blue lane) — annotator text, NOT audio
{fmt_iv(ev.get('narration')) if ev.get('narration') else fmt_list(a.get('said'))}
### Speech (purple lane) — what is actually spoken in the audio (ASR)
{fmt_iv(ev.get('speech')) if ev.get('speech') else '- *(none: ' + str(it.get('speech_layer_note') or 'no speech detected') + ')*'}
### Sounds linked to this object (green lane)
{fmt_iv(ev.get('sound')) if ev.get('sound') else fmt_list(a.get('heard'))}
"""
return md
STATUS_ICON = {"match": "✅", "mismatch": "❌", "missing": "❌", "warn": "⚠️", "info": "ℹ️", "n/a": "➖"}
def crosscheck_md(it):
rows = it.get("crosscheck") or []
n_ok = sum(r["status"] == "match" for r in rows)
n_bad = sum(r["status"] in ("mismatch", "missing") for r in rows)
n_warn = sum(r["status"] == "warn" for r in rows)
md = [f"## Automatic cross-check vs original HD-EPIC annotation — {n_ok} match, {n_bad} mismatch, {n_warn} warn",
"| | what | memory (generated) | original HD-EPIC | note |", "|---|---|---|---|---|"]
for r in rows:
md.append(f"| {STATUS_ICON.get(r['status'], '')} | {r['what']} | {str(r['memory']).replace('|', '/')} | "
f"{str(r['original']).replace('|', '/')} | {r.get('note', '')} |")
md.append("\n### All HD-EPIC narrations in this clip window (✔ = attached to this object)")
for n in it.get("raw_narrations_in_window") or []:
md.append(f"- {'✔' if n['attached'] else '&nbsp;&nbsp;'} **{hms(n['t0'])}–{hms(n['t1'])}** {n['text']}")
return "\n".join(md)
return md
def clock_banner(it):
"""Big on-page clock above the video; a script keeps it in sync with playback."""
t0 = it["window_sec"][0]
return (f'<div id="vt-banner"><div class="vt-clock">⏱ <b>VIDEO TIME</b>'
f'<span id="live-clock" data-t0="{t0}">{hms(t0)}</span></div>'
f'<div class="vt-note"><b>Every time in the questions refers to this clock</b> '
f'(hours : minutes : seconds of the recording; also in yellow at the top-right of the video).<br>'
f'This clip runs {hms(t0, 0)} – {hms(it["window_sec"][1], 0)}.</div></div>')
CLOCK_JS = """
() => {
const fmt = (t) => {
t = Math.round(t * 10) / 10;
const h = Math.floor(t / 3600), m = Math.floor((t % 3600) / 60), s = t - h * 3600 - m * 60;
return h + ":" + String(m).padStart(2, "0") + ":" + (s < 10 ? "0" : "") + s.toFixed(1);
};
setInterval(() => {
const v = document.querySelector("#eval-video video"), c = document.getElementById("live-clock");
if (v && c) c.textContent = fmt(parseFloat(c.dataset.t0) + (v.currentTime || 0));
}, 100);
}
"""
def gallery_for(it):
"""Keyframe panels as plain <img> tags (full width, stacked) -- the Gallery component
crops wide images. Files are served by Gradio from the allowed media directory."""
out = []
for k in it["keyframes"]:
url = f"/gradio_api/file={MEDIA / k['path']}"
out.append(f'<a href="{url}" target="_blank"><img src="{url}" alt="keyframe at {hms(k["t"])}" '
f'style="width:100%;display:block;margin:0 0 8px 0;border-radius:6px"></a>')
return "".join(out) or "<i>no keyframe</i>"
def show(idx, annotator, version):
"""idx is the position inside the chosen version's item list."""
ix = VERSION_ITEMS[version]
idx = max(0, min(idx, len(ix) - 1))
it = ITEMS[ix[idx]]
done = load_done(annotator, version) if annotator else set()
status = "✅ answered" if item_key(it) in done else "⬜ not answered"
header = (f"### {annotator or '?'} · version {version} — item {idx + 1} / {len(ix)} &nbsp; {status} &nbsp; "
f"({len(done)} / {len(ix)} done)")
return ([idx, header, clock_banner(it), str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)]
+ dim_questions(it) + [None] * len(DIMENSIONS) + narration_updates(it) + [""])
def first_unanswered(annotator, version):
done = load_done(annotator, version)
for i, k in enumerate(VERSION_ITEMS[version]):
if item_key(ITEMS[k]) not in done:
return i
return 0
# ---------------------------------------------------------------- app
# Side-by-side layout: the video stays put on the left while the questions scroll on the
# right, so nobody has to scroll away from the clip to answer. Stacks again on narrow screens.
CSS = """
#eval-row { align-items: flex-start; flex-wrap: nowrap; }
#left-pane, #right-pane { max-height: calc(100vh - 70px); overflow-y: auto; flex-wrap: nowrap; }
/* a column is a flex box: without this its children are squeezed to fit instead of scrolling */
#left-pane > *, #right-pane > * { flex-shrink: 0; }
#right-pane { padding-right: 10px; }
#eval-video video { max-height: 56vh; width: 100%; object-fit: contain; }
#kf-gallery img { cursor: zoom-in; }
#vt-banner { background: #ffe14d; color: #111; border: 2px solid #111; border-radius: 8px; padding: 8px 12px;
display: flex; align-items: center; gap: 14px; flex-wrap: wrap; }
#vt-banner .vt-clock { display: flex; align-items: center; gap: 8px; font-size: 15px; white-space: nowrap; }
#vt-banner #live-clock { font: 700 30px/1 ui-monospace, Menlo, Consolas, monospace;
background: #111; color: #ffe14d; padding: 5px 10px; border-radius: 6px; }
#vt-banner .vt-note { font-size: 13px; line-height: 1.35; flex: 1 1 240px; color: #111; }
#vt-banner b { color: #111; }
.q-stem { margin-top: 14px !important; margin-bottom: 2px !important; }
.q-stem p, .q-stem li { font-size: 15px; line-height: 1.35; }
.q-ans { margin-bottom: 6px !important; }
@media (max-width: 1000px) {
#eval-row { flex-wrap: wrap; }
#left-pane, #right-pane { max-height: none; overflow-y: visible; }
}
"""
with gr.Blocks(title="Object-centric annotation eval") as demo:
idx_state = gr.State(0)
name_state = gr.State("")
ver_state = gr.State("A")
with gr.Column(visible=True) as login_col:
gr.Markdown("# Object-centric annotation evaluation\nEnter your name to start (progress is saved per name).")
name_in = gr.Textbox(label="Your name", placeholder="e.g. alice")
ver_in = gr.Radio(VERSIONS, value="A", label="Questionnaire version (the one you were assigned)",
info=" · ".join(f"{v}: {len(VERSION_ITEMS[v])} objects" for v in VERSIONS))
start_btn = gr.Button("Start", variant="primary")
with gr.Column(visible=False) as main_col:
header = gr.Markdown()
with gr.Row(elem_id="eval-row"):
# ---- left: what to look at (stays in view) ----
with gr.Column(scale=5, elem_id="left-pane"):
clock_html = gr.HTML()
video = gr.Video(show_label=False, autoplay=True, loop=True, elem_id="eval-video")
gr.Markdown("**Keyframes** — left: full frame · right: **zoom on the object** "
"(green = mask, red = box, ring = gaze). Click an image to open it full size.")
gallery = gr.HTML(elem_id="kf-gallery")
with gr.Accordion("Legend — what the overlays mean", open=False):
gr.Markdown(
"- **VIDEO TIME** (top-right, boxed) = seconds of the full recording; every annotation time uses it "
"(the player's own 0–30 s counter does not).\n"
"- **green overlay** = the object's mask, tracked through the video (absent while the object is out of view) · "
"**red box** = the annotated box, near each keyframe · **red dot** = where the wearer is looking.\n"
"- yellow **GAZE ON OBJECT** = the memory claims a look at the object.\n"
"- **blue NARRATION** = annotator text (not audio), with [start–end] · **purple SPEECH** = words "
"actually spoken · **green SOUND** = sound events.\n"
"- bottom bar = timeline: lanes sound / narration / speech / dwell, white ticks = narration "
"start/end, red ticks = keyframes, white line = now.")
# ---- right: the questions (scrolls on its own) ----
with gr.Column(scale=6, elem_id="right-pane"):
with gr.Accordion("How to judge — read this once", open=True):
gr.Markdown(
"- Each question states **one claim** of an auto-generated memory about **one object**. Say whether it matches the video.\n"
"- **Correct** / **Partially correct** (right idea, a place / time / detail is off) / **Incorrect** / **Can't tell**.\n"
"- **Times** are **hours:minutes:seconds of the recording** → read the big yellow **VIDEO TIME** clock above the video. Differences under ~1 s are fine.\n"
"- **Places** like “the counter #2”: judge the **kind of place**; ignore the number.\n"
"- **Sound**: 🔊 turn it on (audio is amplified).\n"
"- **Gaze**: the red dot is always shown; the gaze question is only about the **moment right before the hand touches the object**.")
q_mds, radios = [], []
for key, _ in DIMENSIONS:
q_mds.append(gr.Markdown(elem_classes="q-stem"))
radios.append(gr.Radio(CHOICES, show_label=False, container=False, elem_classes="q-ans"))
gr.Markdown("### Narrations — judge **each one** separately (blue subtitles; compare with **VIDEO TIME**)")
narr_mds, narr_radios = [], []
for k in range(NARR_MAX):
narr_mds.append(gr.Markdown(visible=(k == 0), elem_classes="q-stem"))
narr_radios.append(gr.Radio(CHOICES, show_label=False, container=False, visible=(k == 0),
elem_classes="q-ans"))
comment = gr.Textbox(label="Comment (optional)", lines=2)
with gr.Row():
prev_btn = gr.Button("◀ Prev")
skip_btn = gr.Button("Skip ▶")
submit_btn = gr.Button("Submit & Next ▶", variant="primary")
msg = gr.Markdown()
with gr.Accordion("Full memory record of this object (reference)", open=False):
ann_md = gr.Markdown()
with gr.Accordion("Cross-check vs original HD-EPIC annotation (reference)", open=False):
xc_md = gr.Markdown()
outputs = [idx_state, header, clock_html, video, gallery, ann_md, xc_md] + q_mds + radios + narr_mds + narr_radios + [comment]
def start(name, version):
name = (name or "").strip()
if not name:
raise gr.Error("Please enter your name.")
if version not in VERSIONS:
raise gr.Error("Please choose a questionnaire version.")
i = first_unanswered(name, version)
return [name, version, gr.update(visible=False), gr.update(visible=True)] + show(i, name, version)
start_btn.click(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs)
name_in.submit(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs)
def preselect(request: gr.Request):
"""A link like .../?v=B pre-selects that version."""
v = (request.query_params.get("v") or "").upper() if request else ""
return gr.update(value=v) if v in VERSIONS else gr.update()
demo.load(preselect, None, [ver_in])
demo.load(None, None, None, js=CLOCK_JS)
prev_btn.click(lambda i, n, v: show(i - 1, n, v), [idx_state, name_state, ver_state], outputs)
skip_btn.click(lambda i, n, v: show(i + 1, n, v), [idx_state, name_state, ver_state], outputs)
def submit(i, name, version, *vals):
answers, narr_ans, cmt = vals[:len(DIMENSIONS)], vals[len(DIMENSIONS):-1], vals[-1]
ix = VERSION_ITEMS[version]
it = ITEMS[ix[i]]
ns = narrations_of(it)
need = max(1, len(ns))
if any(a is None for a in answers) or any(a is None for a in narr_ans[:need]):
raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions and all {need} narration question(s) before submitting.")
row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name, "version": version,
"video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"],
"comment": (cmt or "").strip()}
row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)})
row["narration_ids"] = ";".join(str(e.get("id")) for e in ns) or "(none)"
row.update({f"narration_{k + 1}": (narr_ans[k] if k < need else "") for k in range(NARR_MAX)})
append_row(row, version)
if i + 1 >= len(ix):
return show(i, name, version) + [f"**Saved.** That was the last item of version {version} — all {len(ix)} done. Thank you!"]
return show(i + 1, name, version) + [f"Saved `{it['object_id']}` → {resp_csv(version).name}"]
submit_btn.click(submit, [idx_state, name_state, ver_state] + radios + narr_radios + [comment], outputs + [msg])
if __name__ == "__main__":
ap = argparse.ArgumentParser()
ap.add_argument("--port", type=int, default=7860)
ap.add_argument("--host", default="0.0.0.0")
ap.add_argument("--share", action="store_true")
a = ap.parse_args()
demo.launch(server_name=a.host, server_port=a.port, share=a.share, allowed_paths=[str(MEDIA)], show_error=True, css=CSS)