Spaces:
Runtime error
Runtime error
File size: 29,117 Bytes
619c2d0 c71f3e6 619c2d0 a4222ff 619c2d0 a333d52 619c2d0 a4222ff a333d52 a4222ff a333d52 a4222ff 41642a1 c71f3e6 41642a1 c71f3e6 41642a1 c71f3e6 41642a1 c71f3e6 41642a1 c71f3e6 5ff6bdb 1a3b1b1 41642a1 1a3b1b1 41642a1 1a3b1b1 5ff6bdb 1a3b1b1 5ff6bdb 1a3b1b1 5ff6bdb 1a3b1b1 a359c5e 1a3b1b1 5ff6bdb 1a3b1b1 41642a1 1a3b1b1 41642a1 1a3b1b1 41642a1 1a3b1b1 5ff6bdb a4222ff 1a3b1b1 a4222ff 1a3b1b1 a4222ff 41642a1 1a3b1b1 a4222ff 1a3b1b1 a4222ff 1a3b1b1 619c2d0 a333d52 619c2d0 a333d52 619c2d0 a333d52 619c2d0 a333d52 619c2d0 a333d52 619c2d0 a333d52 a359c5e 619c2d0 41642a1 619c2d0 41642a1 619c2d0 645ac50 41642a1 c71f3e6 41642a1 619c2d0 c71f3e6 619c2d0 c71f3e6 619c2d0 c71f3e6 619c2d0 41642a1 619c2d0 41642a1 619c2d0 b01534f 41642a1 b01534f 619c2d0 a333d52 619c2d0 a333d52 41642a1 1a3b1b1 619c2d0 a333d52 619c2d0 ee87b42 b01534f 41642a1 ee87b42 619c2d0 a333d52 619c2d0 a333d52 619c2d0 ee87b42 41642a1 ee87b42 b01534f ee87b42 a359c5e ee87b42 41642a1 ee87b42 619c2d0 41642a1 619c2d0 a333d52 619c2d0 a333d52 619c2d0 a333d52 41642a1 619c2d0 a333d52 619c2d0 a333d52 a4222ff a333d52 a4222ff a333d52 619c2d0 a4222ff a333d52 619c2d0 a333d52 619c2d0 ee87b42 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 | #!/usr/bin/env python
"""Gradio site for human evaluation of object-centric annotations.
Run: python app.py [--port 7860] [--share]
Responses are appended to responses/responses.csv.
"""
import argparse
import csv
import json
import os
import re
from datetime import datetime
from pathlib import Path
import gradio as gr
from filelock import FileLock
HERE = Path(__file__).resolve().parent
os.environ.setdefault("GRADIO_TEMP_DIR", str(HERE / ".gradio_cache"))
MEDIA = HERE / "media"
RESP_DIR = HERE / "responses"
RESP_CSV = RESP_DIR / "responses.csv"
LOCK = HERE / ".responses.csv.lock"
CHOICES = ["Correct", "Partially correct", "Incorrect", "Can't tell"]
# (column name, question shown to the annotator)
DIMENSIONS = [
("identity", "1. Identity β does `category` + `looks like` describe the object inside the red box?"),
("localization", "2. Localization β do the red box / green mask cover the right object (not a neighbour, hand, or background)?"),
("state_history", "3. Location & state history β is the where / relation / timing (`now`, `history`) correct?"),
("moves", "4. Moves β are the recorded movement events (from β to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"),
("sound_speech", "5. Sound & speech β do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"),
("gaze", "6. Gaze β is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"),
]
# ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ----
DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses
SCHEDULER = None
if DATASET_REPO and os.environ.get("HF_TOKEN"):
from huggingface_hub import CommitScheduler, HfApi, hf_hub_download
RESP_DIR.mkdir(exist_ok=True)
try: # resume from the copy in the dataset
api = HfApi(token=os.environ["HF_TOKEN"])
api.create_repo(DATASET_REPO, repo_type="dataset", private=True, exist_ok=True)
except Exception as e:
print("could not reach the responses dataset:", type(e).__name__, str(e)[:120])
for _v in ("A", "B", "C"): # one file per questionnaire version
try:
cached = hf_hub_download(DATASET_REPO, f"responses_{_v}.csv", repo_type="dataset",
token=os.environ["HF_TOKEN"])
(RESP_DIR / f"responses_{_v}.csv").write_bytes(Path(cached).read_bytes())
print(f"restored version {_v}: {sum(1 for _ in open(cached)) - 1} responses")
except Exception as e: # first run: nothing to restore
print(f"version {_v}: nothing restored ({type(e).__name__})")
SCHEDULER = CommitScheduler(repo_id=DATASET_REPO, repo_type="dataset", folder_path=RESP_DIR,
path_in_repo=".", every=2, private=True, token=os.environ["HF_TOKEN"],
allow_patterns=["*.csv"])
ITEMS = json.load(open(MEDIA / "index.json"))
# Narrations are judged one by one. An object has between 0 and NARR_MAX of them, so the
# page holds NARR_MAX radio groups and shows only as many as the current object needs.
NARR_MAX = max(1, max(len(it["events"]["narration"]) for it in ITEMS))
COLUMNS = (["timestamp", "annotator", "version", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS]
+ ["narration_ids"] + [f"narration_{k + 1}" for k in range(NARR_MAX)] + ["comment"])
NO_NARR_Q = ("No narration is linked to this object. Is that right β is there really no narration in this clip "
"about it? (All narrations of the clip are listed in the cross-check panel.)")
# ---- three shorter questionnaires --------------------------------------------------------
# 95 objects is too long for one sitting, so the clips are dealt into versions A / B / C
# (whole clips, so an annotator judges every object of the clips they watch; balanced by
# object count and spread over the videos). Each version writes its own CSV.
VERSIONS = ["A", "B", "C"]
def _split_versions():
clips = {}
for i, it in enumerate(ITEMS):
clips.setdefault((it["video_id"], it["chunk"]), []).append(i)
load = {v: 0 for v in VERSIONS}
out = {v: [] for v in VERSIONS}
by_video = {}
for key in clips:
by_video.setdefault(key[0], []).append(key)
for vid in sorted(by_video): # within a video, biggest clip first -> lightest version
taken = set()
for key in sorted(by_video[vid], key=lambda k: (-len(clips[k]), k)):
cand = [v for v in VERSIONS if v not in taken] or VERSIONS
v = min(cand, key=lambda x: (load[x], x))
taken.add(v)
out[v] += clips[key]
load[v] += len(clips[key])
return {v: sorted(ix) for v, ix in out.items()}
VERSION_ITEMS = _split_versions()
def resp_csv(version):
return RESP_DIR / f"responses_{version}.csv"
# A responses file written with a different column set (older form) is kept under another
# name instead of being appended to with misaligned columns.
for _p in [RESP_CSV] + [resp_csv(v) for v in VERSIONS]:
if _p.exists():
with open(_p, newline="") as _f:
_hdr = next(csv.reader(_f), None)
if _hdr != COLUMNS or _p == RESP_CSV:
_p.rename(RESP_DIR / f"{_p.stem}_old_{datetime.now():%Y%m%d_%H%M%S}.csv")
def hms(t, dec=1):
"""Seconds of the full recording -> H:MM:SS.s (the unit shown everywhere on the site)."""
if t is None:
return "?"
t = round(float(t), dec)
h, m = int(t // 3600), int(t % 3600 // 60)
s = t - h * 3600 - m * 60
return f"{h}:{m:02d}:{s:0{3 + dec if dec else 2}.{dec}f}"
def place(anchor):
"""`P02_counter.002` -> `the counter #2`; `person_01` -> `the wearer's hand`."""
if not anchor or anchor == "unknown":
return "an unknown place"
if anchor.startswith("person"):
return "the wearer's hand"
kind, _, num = re.sub(r"^P\d\d_", "", anchor).partition(".")
return f"the {kind.replace('_', ' ')}" + (f" #{int(num)}" if num.isdigit() else "")
def where(rel, anchor):
if rel == "held_by":
return "held in the wearer's hand"
return {"inside": "inside", "on": "on", "hanging_on": "hanging on"}.get(rel, "at") + " " + place(anchor)
def state_sentence(st):
a, b = st["interval_sec"]
w = where(st["relation"], st["anchor"])
if b is None:
return f"from {hms(a)} until the end of the clip: {w}"
if abs(b - a) < 0.05:
return f"at {hms(a)}: {w}"
return f"from {hms(a)} to {hms(b)}: {w}"
def move_sentence(tr):
a, b = tr["interval_sec"]
when = f"at {hms(a)}" if abs(b - a) < 0.05 else f"between {hms(a)} and {hms(b)}"
return (f"{when} it goes from β{where(tr['from']['relation'], tr['from']['anchor'])}β to "
f"β{where(tr['to']['relation'], tr['to']['anchor'])}β (action: {tr['event'].replace('_', ' ')})")
def gaze_summary(it):
a, att = it["annotation"], it.get("attention") or {}
if a.get("looked_at"):
gap = att.get("prime_gap_sec")
return (f"Yes β first look at {hms(att.get('first_gaze_sec'))}"
+ (f", about {gap:.1f}s before the hand touches it" if isinstance(gap, (int, float)) else ""))
return "No β the memory found no look at this object before the hand touches it"
def _join(xs):
return "; ".join(f"({i + 1}) {x}" for i, x in enumerate(xs))
def _bul(xs):
return "\n".join(f"- {x}" for x in xs)
def state_md(st):
a, b = st["interval_sec"]
w = f"**{where(st['relation'], st['anchor'])}**"
if b is None:
return f"{hms(a)} β end of clip: {w}"
return f"at {hms(a)}: {w}" if abs(b - a) < 0.05 else f"{hms(a)} β {hms(b)}: {w}"
def move_md(tr):
a, b = tr["interval_sec"]
when = f"at **{hms(a)}**" if abs(b - a) < 0.05 else f"**{hms(a)} β {hms(b)}**"
return (f"{when}: {where(tr['from']['relation'], tr['from']['anchor'])} **β** "
f"{where(tr['to']['relation'], tr['to']['anchor'])} ({tr['event'].replace('_', ' ')})")
def dim_questions(it):
"""Markdown text of the six fixed questions for THIS object: short, claim in bold."""
a, ev, att = it["annotation"], it.get("events") or {}, it.get("attention") or {}
obj = f"**{a.get('category')}**"
q = {}
q["identity"] = f"**1. Identity** β is the boxed object a {obj} that looks like β**{a.get('looks_like')}**β?"
q["localization"] = f"**2. Box / mask** β does the **green mask** follow the {obj} **through the video**, and do the **red box** and mask cover it in the **keyframes (zoom on the right)**? (not a neighbour, the hand or background)"
states = it.get("states") or []
q["state_history"] = (f"**3. Location** β is the {obj} where the memory says, at these times?\n"
+ _bul([state_md(x) for x in states] or (a.get("history") or [])))
moves = it.get("transitions") or []
q["moves"] = ((f"**4. Moves** *(move = picked up / put down / put into / taken out)* β are these **right**, "
f"and is **none missing**?\n" + _bul([move_md(x) for x in moves]))
if moves else
f"**4. Moves** *(move = picked up / put down / put into / taken out)* β the memory records "
f"**no move**. Does the {obj} really **stay in place** for the whole clip?")
snd = [f"**{e['label']}** at {hms(e['t0'])}β{hms(e['t1'])}" for e in ev.get("sound") or []]
sp = [f"β{e['label']}β at {hms(e['t0'])}β{hms(e['t1'])}" for e in ev.get("speech") or []]
q["sound_speech"] = ((f"**5. Sounds π (turn sound on)** β can you **hear** each one at that time, and is it made "
f"**by / with the** {obj}?\n" + _bul(snd))
if snd else
f"**5. Sounds π (turn sound on)** β the memory links **no sound**. Does the {obj} really "
"make **no sound** in the clip?")
if sp:
q["sound_speech"] += "\n\nSpeech in the clip β transcribed correctly?\n" + _bul(sp)
HEAD = ("**6. Gaze right before touching** β only the **moment right before the hand touches the object** "
"counts, **not the whole clip** (the red dot is on screen all the time).\n\n")
fg, gap = att.get("first_gaze_sec"), att.get("prime_gap_sec")
t0, t1 = it["window_sec"]
if not a.get("looked_at"):
q["gaze"] = (HEAD + f"Memory: the wearer did **NOT look** at the {obj} before touching it. "
"In the **2β3 s before the hand touches it**, does the red dot **stay off** the object?")
elif not (t0 <= (fg or -1) <= t1):
q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **{hms(fg)}**, which is **outside this clip** "
f"({hms(t0, 0)}β{hms(t1, 0)}) β please answer **Can't tell**.")
else:
q["gaze"] = (HEAD + f"Memory: they looked at the {obj} at **β {hms(fg)}**"
+ (f" (**{gap:.1f}s before touching it**)" if isinstance(gap, (int, float)) else "")
+ ". **At that moment**, is the **red dot on (or right next to) the object**?")
return [q[k] for k, _ in DIMENSIONS]
def narrations_of(it):
return sorted(it["events"]["narration"], key=lambda e: e["t0"])
def narration_updates(it):
"""For each of the NARR_MAX slots: (markdown update, radio update) for this object."""
ns = narrations_of(it)
mds, rds = [], []
for k in range(NARR_MAX):
if k < len(ns):
e = ns[k]
mds.append(gr.update(visible=True, value=f"**Narration {k + 1} / {len(ns)}** Β· **{hms(e['t0'])}β{hms(e['t1'])}** Β· "
f"β{e['label']}β β about **this object**, at the **right time**?"))
rds.append(gr.update(visible=True, value=None))
elif k == 0:
mds.append(gr.update(visible=True, value="**No narration is linked** to this object. Is that right β does **no "
"narration in this clip** talk about it? (all narrations: cross-check panel)"))
rds.append(gr.update(visible=True, value=None))
else:
mds.append(gr.update(visible=False, value=""))
rds.append(gr.update(visible=False, value=None))
return mds + rds
N = len(ITEMS)
def item_key(it):
return (it["video_id"], it["chunk"], it["object_id"])
# ---------------------------------------------------------------- csv i/o
def load_done(annotator, version):
done = set()
p = resp_csv(version)
if p.exists():
with open(p, newline="") as f:
for r in csv.DictReader(f):
if r["annotator"] == annotator:
done.add((r["video_id"], r["chunk"], r["object_id"]))
return done
def _write(row, version):
p = resp_csv(version)
new = not p.exists()
with open(p, "a", newline="") as f:
w = csv.DictWriter(f, fieldnames=COLUMNS)
if new:
w.writeheader()
w.writerow(row)
def append_row(row, version):
RESP_DIR.mkdir(exist_ok=True)
with FileLock(str(LOCK)):
if SCHEDULER is not None:
with SCHEDULER.lock:
_write(row, version)
else:
_write(row, version)
if SCHEDULER is not None:
# Push right away instead of waiting for the 2-minute tick: a restart of the Space
# (every redeploy causes one) would otherwise lose whatever was not yet synced.
try:
SCHEDULER.trigger()
except Exception as e:
print("immediate sync failed:", type(e).__name__, str(e)[:120])
# ---------------------------------------------------------------- rendering
def fmt_list(xs):
return "\n".join(f"- {x}" for x in xs) if xs else "- *(none)*"
def fmt_iv(evs):
return "\n".join(f"- **{hms(e['t0'])}β{hms(e['t1'])}** {e['label']}" for e in evs) if evs else "- *(none)*"
def annotation_md(it):
a = it["annotation"]
now = a.get("now") or {}
lc = it.get("lifecycle") or {}
ev = it.get("events") or {}
att = it.get("attention") or {}
md = f"""## `{it['object_id']}`
**video** `{it['video_id']}` / {it['chunk']} window {hms(it['window_sec'][0])}β{hms(it['window_sec'][1])}
| field | value |
|---|---|
| **category** | {a.get('category')} |
| **looks like** | {a.get('looks_like')} |
| **color / material** | {a.get('color', '?')} / {a.get('material', '?')} |
| **at the end of the clip** | {where(now.get('relation'), now.get('at'))} (since {hms(now.get('since_sec'))}){(' Β· contains: ' + str(now['contains'])) if now.get('contains') else ''} |
| **looked at before touching** | {gaze_summary(it)} |
| **lifecycle** | first {hms(lc.get('first_observed_sec'))} β last {hms(lc.get('last_observed_sec'))}, {lc.get('final_status')} |
| **confidence** | {it.get('confidence')} |
### Where it is (location history)
{fmt_list([state_sentence(x) for x in it.get('states') or []] or a.get('history'))}
### Moves (changes of place)
{fmt_list([move_sentence(x) for x in it.get('transitions') or []])}
### Claimed look at the object before touching it (yellow lane; the red dot itself is always shown)
{fmt_iv(ev.get('dwell')) if a.get('looked_at') else '- *(none claimed)*'}
### Narrations linked to this object (blue lane) β annotator text, NOT audio
{fmt_iv(ev.get('narration')) if ev.get('narration') else fmt_list(a.get('said'))}
### Speech (purple lane) β what is actually spoken in the audio (ASR)
{fmt_iv(ev.get('speech')) if ev.get('speech') else '- *(none: ' + str(it.get('speech_layer_note') or 'no speech detected') + ')*'}
### Sounds linked to this object (green lane)
{fmt_iv(ev.get('sound')) if ev.get('sound') else fmt_list(a.get('heard'))}
"""
return md
STATUS_ICON = {"match": "β
", "mismatch": "β", "missing": "β", "warn": "β οΈ", "info": "βΉοΈ", "n/a": "β"}
def crosscheck_md(it):
rows = it.get("crosscheck") or []
n_ok = sum(r["status"] == "match" for r in rows)
n_bad = sum(r["status"] in ("mismatch", "missing") for r in rows)
n_warn = sum(r["status"] == "warn" for r in rows)
md = [f"## Automatic cross-check vs original HD-EPIC annotation β {n_ok} match, {n_bad} mismatch, {n_warn} warn",
"| | what | memory (generated) | original HD-EPIC | note |", "|---|---|---|---|---|"]
for r in rows:
md.append(f"| {STATUS_ICON.get(r['status'], '')} | {r['what']} | {str(r['memory']).replace('|', '/')} | "
f"{str(r['original']).replace('|', '/')} | {r.get('note', '')} |")
md.append("\n### All HD-EPIC narrations in this clip window (β = attached to this object)")
for n in it.get("raw_narrations_in_window") or []:
md.append(f"- {'β' if n['attached'] else ' '} **{hms(n['t0'])}β{hms(n['t1'])}** {n['text']}")
return "\n".join(md)
return md
def clock_banner(it):
"""Big on-page clock above the video; a script keeps it in sync with playback."""
t0 = it["window_sec"][0]
return (f'<div id="vt-banner"><div class="vt-clock">β± <b>VIDEO TIME</b>'
f'<span id="live-clock" data-t0="{t0}">{hms(t0)}</span></div>'
f'<div class="vt-note"><b>Every time in the questions refers to this clock</b> '
f'(hours : minutes : seconds of the recording; also in yellow at the top-right of the video).<br>'
f'This clip runs {hms(t0, 0)} β {hms(it["window_sec"][1], 0)}.</div></div>')
CLOCK_JS = """
() => {
const fmt = (t) => {
t = Math.round(t * 10) / 10;
const h = Math.floor(t / 3600), m = Math.floor((t % 3600) / 60), s = t - h * 3600 - m * 60;
return h + ":" + String(m).padStart(2, "0") + ":" + (s < 10 ? "0" : "") + s.toFixed(1);
};
setInterval(() => {
const v = document.querySelector("#eval-video video"), c = document.getElementById("live-clock");
if (v && c) c.textContent = fmt(parseFloat(c.dataset.t0) + (v.currentTime || 0));
}, 100);
}
"""
def gallery_for(it):
"""Keyframe panels as plain <img> tags (full width, stacked) -- the Gallery component
crops wide images. Files are served by Gradio from the allowed media directory."""
out = []
for k in it["keyframes"]:
url = f"/gradio_api/file={MEDIA / k['path']}"
out.append(f'<a href="{url}" target="_blank"><img src="{url}" alt="keyframe at {hms(k["t"])}" '
f'style="width:100%;display:block;margin:0 0 8px 0;border-radius:6px"></a>')
return "".join(out) or "<i>no keyframe</i>"
def show(idx, annotator, version):
"""idx is the position inside the chosen version's item list."""
ix = VERSION_ITEMS[version]
idx = max(0, min(idx, len(ix) - 1))
it = ITEMS[ix[idx]]
done = load_done(annotator, version) if annotator else set()
status = "β
answered" if item_key(it) in done else "β¬ not answered"
header = (f"### {annotator or '?'} Β· version {version} β item {idx + 1} / {len(ix)} {status} "
f"({len(done)} / {len(ix)} done)")
return ([idx, header, clock_banner(it), str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)]
+ dim_questions(it) + [None] * len(DIMENSIONS) + narration_updates(it) + [""])
def first_unanswered(annotator, version):
done = load_done(annotator, version)
for i, k in enumerate(VERSION_ITEMS[version]):
if item_key(ITEMS[k]) not in done:
return i
return 0
# ---------------------------------------------------------------- app
# Side-by-side layout: the video stays put on the left while the questions scroll on the
# right, so nobody has to scroll away from the clip to answer. Stacks again on narrow screens.
CSS = """
#eval-row { align-items: flex-start; flex-wrap: nowrap; }
#left-pane, #right-pane { max-height: calc(100vh - 70px); overflow-y: auto; flex-wrap: nowrap; }
/* a column is a flex box: without this its children are squeezed to fit instead of scrolling */
#left-pane > *, #right-pane > * { flex-shrink: 0; }
#right-pane { padding-right: 10px; }
#eval-video video { max-height: 56vh; width: 100%; object-fit: contain; }
#kf-gallery img { cursor: zoom-in; }
#vt-banner { background: #ffe14d; color: #111; border: 2px solid #111; border-radius: 8px; padding: 8px 12px;
display: flex; align-items: center; gap: 14px; flex-wrap: wrap; }
#vt-banner .vt-clock { display: flex; align-items: center; gap: 8px; font-size: 15px; white-space: nowrap; }
#vt-banner #live-clock { font: 700 30px/1 ui-monospace, Menlo, Consolas, monospace;
background: #111; color: #ffe14d; padding: 5px 10px; border-radius: 6px; }
#vt-banner .vt-note { font-size: 13px; line-height: 1.35; flex: 1 1 240px; color: #111; }
#vt-banner b { color: #111; }
.q-stem { margin-top: 14px !important; margin-bottom: 2px !important; }
.q-stem p, .q-stem li { font-size: 15px; line-height: 1.35; }
.q-ans { margin-bottom: 6px !important; }
@media (max-width: 1000px) {
#eval-row { flex-wrap: wrap; }
#left-pane, #right-pane { max-height: none; overflow-y: visible; }
}
"""
with gr.Blocks(title="Object-centric annotation eval") as demo:
idx_state = gr.State(0)
name_state = gr.State("")
ver_state = gr.State("A")
with gr.Column(visible=True) as login_col:
gr.Markdown("# Object-centric annotation evaluation\nEnter your name to start (progress is saved per name).")
name_in = gr.Textbox(label="Your name", placeholder="e.g. alice")
ver_in = gr.Radio(VERSIONS, value="A", label="Questionnaire version (the one you were assigned)",
info=" Β· ".join(f"{v}: {len(VERSION_ITEMS[v])} objects" for v in VERSIONS))
start_btn = gr.Button("Start", variant="primary")
with gr.Column(visible=False) as main_col:
header = gr.Markdown()
with gr.Row(elem_id="eval-row"):
# ---- left: what to look at (stays in view) ----
with gr.Column(scale=5, elem_id="left-pane"):
clock_html = gr.HTML()
video = gr.Video(show_label=False, autoplay=True, loop=True, elem_id="eval-video")
gr.Markdown("**Keyframes** β left: full frame Β· right: **zoom on the object** "
"(green = mask, red = box, ring = gaze). Click an image to open it full size.")
gallery = gr.HTML(elem_id="kf-gallery")
with gr.Accordion("Legend β what the overlays mean", open=False):
gr.Markdown(
"- **VIDEO TIME** (top-right, boxed) = seconds of the full recording; every annotation time uses it "
"(the player's own 0β30 s counter does not).\n"
"- **green overlay** = the object's mask, tracked through the video (absent while the object is out of view) Β· "
"**red box** = the annotated box, near each keyframe Β· **red dot** = where the wearer is looking.\n"
"- yellow **GAZE ON OBJECT** = the memory claims a look at the object.\n"
"- **blue NARRATION** = annotator text (not audio), with [startβend] Β· **purple SPEECH** = words "
"actually spoken Β· **green SOUND** = sound events.\n"
"- bottom bar = timeline: lanes sound / narration / speech / dwell, white ticks = narration "
"start/end, red ticks = keyframes, white line = now.")
# ---- right: the questions (scrolls on its own) ----
with gr.Column(scale=6, elem_id="right-pane"):
with gr.Accordion("How to judge β read this once", open=True):
gr.Markdown(
"- Each question states **one claim** of an auto-generated memory about **one object**. Say whether it matches the video.\n"
"- **Correct** / **Partially correct** (right idea, a place / time / detail is off) / **Incorrect** / **Can't tell**.\n"
"- **Times** are **hours:minutes:seconds of the recording** β read the big yellow **VIDEO TIME** clock above the video. Differences under ~1 s are fine.\n"
"- **Places** like βthe counter #2β: judge the **kind of place**; ignore the number.\n"
"- **Sound**: π turn it on (audio is amplified).\n"
"- **Gaze**: the red dot is always shown; the gaze question is only about the **moment right before the hand touches the object**.")
q_mds, radios = [], []
for key, _ in DIMENSIONS:
q_mds.append(gr.Markdown(elem_classes="q-stem"))
radios.append(gr.Radio(CHOICES, show_label=False, container=False, elem_classes="q-ans"))
gr.Markdown("### Narrations β judge **each one** separately (blue subtitles; compare with **VIDEO TIME**)")
narr_mds, narr_radios = [], []
for k in range(NARR_MAX):
narr_mds.append(gr.Markdown(visible=(k == 0), elem_classes="q-stem"))
narr_radios.append(gr.Radio(CHOICES, show_label=False, container=False, visible=(k == 0),
elem_classes="q-ans"))
comment = gr.Textbox(label="Comment (optional)", lines=2)
with gr.Row():
prev_btn = gr.Button("β Prev")
skip_btn = gr.Button("Skip βΆ")
submit_btn = gr.Button("Submit & Next βΆ", variant="primary")
msg = gr.Markdown()
with gr.Accordion("Full memory record of this object (reference)", open=False):
ann_md = gr.Markdown()
with gr.Accordion("Cross-check vs original HD-EPIC annotation (reference)", open=False):
xc_md = gr.Markdown()
outputs = [idx_state, header, clock_html, video, gallery, ann_md, xc_md] + q_mds + radios + narr_mds + narr_radios + [comment]
def start(name, version):
name = (name or "").strip()
if not name:
raise gr.Error("Please enter your name.")
if version not in VERSIONS:
raise gr.Error("Please choose a questionnaire version.")
i = first_unanswered(name, version)
return [name, version, gr.update(visible=False), gr.update(visible=True)] + show(i, name, version)
start_btn.click(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs)
name_in.submit(start, [name_in, ver_in], [name_state, ver_state, login_col, main_col] + outputs)
def preselect(request: gr.Request):
"""A link like .../?v=B pre-selects that version."""
v = (request.query_params.get("v") or "").upper() if request else ""
return gr.update(value=v) if v in VERSIONS else gr.update()
demo.load(preselect, None, [ver_in])
demo.load(None, None, None, js=CLOCK_JS)
prev_btn.click(lambda i, n, v: show(i - 1, n, v), [idx_state, name_state, ver_state], outputs)
skip_btn.click(lambda i, n, v: show(i + 1, n, v), [idx_state, name_state, ver_state], outputs)
def submit(i, name, version, *vals):
answers, narr_ans, cmt = vals[:len(DIMENSIONS)], vals[len(DIMENSIONS):-1], vals[-1]
ix = VERSION_ITEMS[version]
it = ITEMS[ix[i]]
ns = narrations_of(it)
need = max(1, len(ns))
if any(a is None for a in answers) or any(a is None for a in narr_ans[:need]):
raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions and all {need} narration question(s) before submitting.")
row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name, "version": version,
"video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"],
"comment": (cmt or "").strip()}
row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)})
row["narration_ids"] = ";".join(str(e.get("id")) for e in ns) or "(none)"
row.update({f"narration_{k + 1}": (narr_ans[k] if k < need else "") for k in range(NARR_MAX)})
append_row(row, version)
if i + 1 >= len(ix):
return show(i, name, version) + [f"**Saved.** That was the last item of version {version} β all {len(ix)} done. Thank you!"]
return show(i + 1, name, version) + [f"Saved `{it['object_id']}` β {resp_csv(version).name}"]
submit_btn.click(submit, [idx_state, name_state, ver_state] + radios + narr_radios + [comment], outputs + [msg])
if __name__ == "__main__":
ap = argparse.ArgumentParser()
ap.add_argument("--port", type=int, default=7860)
ap.add_argument("--host", default="0.0.0.0")
ap.add_argument("--share", action="store_true")
a = ap.parse_args()
demo.launch(server_name=a.host, server_port=a.port, share=a.share, allowed_paths=[str(MEDIA)], show_error=True, css=CSS)
|