Spaces:
Runtime error
Runtime error
labelled VIDEO TIME clock + time axis
Browse files
app.py
CHANGED
|
@@ -28,11 +28,9 @@ DIMENSIONS = [
|
|
| 28 |
("localization", "2. Localization β do the red box / green mask cover the right object (not a neighbour, hand, or background)?"),
|
| 29 |
("state_history", "3. Location & state history β is the where / relation / timing (`now`, `history`) correct?"),
|
| 30 |
("moves", "4. Moves β are the recorded movement events (from β to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"),
|
| 31 |
-
("
|
| 32 |
-
("
|
| 33 |
-
("gaze", "7. Gaze β is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"),
|
| 34 |
]
|
| 35 |
-
COLUMNS = ["timestamp", "annotator", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS] + ["comment"]
|
| 36 |
|
| 37 |
# ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ----
|
| 38 |
DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses
|
|
@@ -53,6 +51,42 @@ if DATASET_REPO and os.environ.get("HF_TOKEN"):
|
|
| 53 |
allow_patterns=["*.csv"])
|
| 54 |
|
| 55 |
ITEMS = json.load(open(MEDIA / "index.json"))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
N = len(ITEMS)
|
| 57 |
|
| 58 |
|
|
@@ -169,7 +203,8 @@ def show(idx, annotator):
|
|
| 169 |
done = load_done(annotator) if annotator else set()
|
| 170 |
status = "β
answered" if item_key(it) in done else "β¬ not answered"
|
| 171 |
header = f"### {annotator or '?'} β item {idx + 1} / {N} {status} ({len(done)} / {N} done)"
|
| 172 |
-
return [idx, header, str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)]
|
|
|
|
| 173 |
|
| 174 |
|
| 175 |
def first_unanswered(annotator):
|
|
@@ -193,8 +228,9 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
|
|
| 193 |
with gr.Column(visible=False) as main_col:
|
| 194 |
header = gr.Markdown()
|
| 195 |
gr.Markdown(
|
| 196 |
-
"All times are
|
| 197 |
-
"
|
|
|
|
| 198 |
"**Clip overlays:** red box = object near each keyframe Β· red dot = where the wearer is looking (gaze) Β· "
|
| 199 |
"yellow *GAZE ON OBJECT* = the memory claims a gaze dwell Β· **blue NARRATION** = HD-EPIC annotator text "
|
| 200 |
"(not audio) with its [start-end] and START/END markers Β· **purple SPEECH** = words actually spoken in the "
|
|
@@ -212,6 +248,9 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
|
|
| 212 |
xc_md = gr.Markdown()
|
| 213 |
gr.Markdown("## Your judgement")
|
| 214 |
radios = [gr.Radio(CHOICES, label=q) for _, q in DIMENSIONS]
|
|
|
|
|
|
|
|
|
|
| 215 |
comment = gr.Textbox(label="Comment (optional)", lines=2)
|
| 216 |
with gr.Row():
|
| 217 |
prev_btn = gr.Button("β Prev")
|
|
@@ -219,7 +258,7 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
|
|
| 219 |
submit_btn = gr.Button("Submit & Next βΆ", variant="primary")
|
| 220 |
msg = gr.Markdown()
|
| 221 |
|
| 222 |
-
outputs = [idx_state, header, video, gallery, ann_md, xc_md] + radios + [comment]
|
| 223 |
|
| 224 |
def start(name):
|
| 225 |
name = (name or "").strip()
|
|
@@ -235,20 +274,24 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
|
|
| 235 |
skip_btn.click(lambda i, n: show(i + 1, n), [idx_state, name_state], outputs)
|
| 236 |
|
| 237 |
def submit(i, name, *vals):
|
| 238 |
-
|
| 239 |
-
if any(a is None for a in answers):
|
| 240 |
-
raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions before submitting.")
|
| 241 |
it = ITEMS[i]
|
|
|
|
|
|
|
|
|
|
|
|
|
| 242 |
row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name,
|
| 243 |
"video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"],
|
| 244 |
"comment": (cmt or "").strip()}
|
| 245 |
row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)})
|
|
|
|
|
|
|
| 246 |
append_row(row)
|
| 247 |
if i + 1 >= N:
|
| 248 |
return show(i, name) + [f"**Saved.** That was the last item β all {N} done. Thank you!"]
|
| 249 |
return show(i + 1, name) + [f"Saved `{it['object_id']}` β {RESP_CSV.name}"]
|
| 250 |
|
| 251 |
-
submit_btn.click(submit, [idx_state, name_state] + radios + [comment], outputs + [msg])
|
| 252 |
|
| 253 |
if __name__ == "__main__":
|
| 254 |
ap = argparse.ArgumentParser()
|
|
|
|
| 28 |
("localization", "2. Localization β do the red box / green mask cover the right object (not a neighbour, hand, or background)?"),
|
| 29 |
("state_history", "3. Location & state history β is the where / relation / timing (`now`, `history`) correct?"),
|
| 30 |
("moves", "4. Moves β are the recorded movement events (from β to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"),
|
| 31 |
+
("sound_speech", "5. Sound & speech β do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"),
|
| 32 |
+
("gaze", "6. Gaze β is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"),
|
|
|
|
| 33 |
]
|
|
|
|
| 34 |
|
| 35 |
# ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ----
|
| 36 |
DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses
|
|
|
|
| 51 |
allow_patterns=["*.csv"])
|
| 52 |
|
| 53 |
ITEMS = json.load(open(MEDIA / "index.json"))
|
| 54 |
+
# Narrations are judged one by one. An object has between 0 and NARR_MAX of them, so the
|
| 55 |
+
# page holds NARR_MAX radio groups and shows only as many as the current object needs.
|
| 56 |
+
NARR_MAX = max(1, max(len(it["events"]["narration"]) for it in ITEMS))
|
| 57 |
+
COLUMNS = (["timestamp", "annotator", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS]
|
| 58 |
+
+ ["narration_ids"] + [f"narration_{k + 1}" for k in range(NARR_MAX)] + ["comment"])
|
| 59 |
+
NO_NARR_Q = ("No narration is linked to this object. Is that right β is there really no narration in this clip "
|
| 60 |
+
"about it? (All narrations of the clip are listed in the cross-check panel.)")
|
| 61 |
+
|
| 62 |
+
# A responses file written with a different column set (older version of the form) is kept
|
| 63 |
+
# under another name instead of being appended to with misaligned columns.
|
| 64 |
+
if RESP_CSV.exists():
|
| 65 |
+
with open(RESP_CSV, newline="") as _f:
|
| 66 |
+
_hdr = next(csv.reader(_f), None)
|
| 67 |
+
if _hdr != COLUMNS:
|
| 68 |
+
RESP_CSV.rename(RESP_DIR / f"responses_old_{datetime.now():%Y%m%d_%H%M%S}.csv")
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def narrations_of(it):
|
| 72 |
+
return sorted(it["events"]["narration"], key=lambda e: e["t0"])
|
| 73 |
+
|
| 74 |
+
|
| 75 |
+
def narration_updates(it):
|
| 76 |
+
"""gr.update for each of the NARR_MAX radio groups: label + visibility for this object."""
|
| 77 |
+
ns = narrations_of(it)
|
| 78 |
+
ups = []
|
| 79 |
+
for k in range(NARR_MAX):
|
| 80 |
+
if k < len(ns):
|
| 81 |
+
e = ns[k]
|
| 82 |
+
ups.append(gr.update(visible=True, value=None,
|
| 83 |
+
label=f"Narration {k + 1} of {len(ns)} β [{e['t0']:.1f}β{e['t1']:.1f}s] β{e['label']}β β "
|
| 84 |
+
"is this about THIS object, and is the time right?"))
|
| 85 |
+
elif k == 0:
|
| 86 |
+
ups.append(gr.update(visible=True, value=None, label=NO_NARR_Q))
|
| 87 |
+
else:
|
| 88 |
+
ups.append(gr.update(visible=False, value=None))
|
| 89 |
+
return ups
|
| 90 |
N = len(ITEMS)
|
| 91 |
|
| 92 |
|
|
|
|
| 203 |
done = load_done(annotator) if annotator else set()
|
| 204 |
status = "β
answered" if item_key(it) in done else "β¬ not answered"
|
| 205 |
header = f"### {annotator or '?'} β item {idx + 1} / {N} {status} ({len(done)} / {N} done)"
|
| 206 |
+
return ([idx, header, str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)]
|
| 207 |
+
+ [None] * len(DIMENSIONS) + narration_updates(it) + [""])
|
| 208 |
|
| 209 |
|
| 210 |
def first_unanswered(annotator):
|
|
|
|
| 228 |
with gr.Column(visible=False) as main_col:
|
| 229 |
header = gr.Markdown()
|
| 230 |
gr.Markdown(
|
| 231 |
+
"β± **All annotation times are seconds in the full recording.** The current time is the boxed "
|
| 232 |
+
"**VIDEO TIME** at the top-right of the video and the white playhead on the bottom time axis "
|
| 233 |
+
"(the player's own 0β30 s counter is NOT the annotation time). "
|
| 234 |
"**Clip overlays:** red box = object near each keyframe Β· red dot = where the wearer is looking (gaze) Β· "
|
| 235 |
"yellow *GAZE ON OBJECT* = the memory claims a gaze dwell Β· **blue NARRATION** = HD-EPIC annotator text "
|
| 236 |
"(not audio) with its [start-end] and START/END markers Β· **purple SPEECH** = words actually spoken in the "
|
|
|
|
| 248 |
xc_md = gr.Markdown()
|
| 249 |
gr.Markdown("## Your judgement")
|
| 250 |
radios = [gr.Radio(CHOICES, label=q) for _, q in DIMENSIONS]
|
| 251 |
+
gr.Markdown("### Narrations β judge each one separately\n"
|
| 252 |
+
"Blue NARRATION subtitles in the video; compare the [startβend] with the VIDEO TIME clock.")
|
| 253 |
+
narr_radios = [gr.Radio(CHOICES, label=f"Narration {k + 1}", visible=(k == 0)) for k in range(NARR_MAX)]
|
| 254 |
comment = gr.Textbox(label="Comment (optional)", lines=2)
|
| 255 |
with gr.Row():
|
| 256 |
prev_btn = gr.Button("β Prev")
|
|
|
|
| 258 |
submit_btn = gr.Button("Submit & Next βΆ", variant="primary")
|
| 259 |
msg = gr.Markdown()
|
| 260 |
|
| 261 |
+
outputs = [idx_state, header, video, gallery, ann_md, xc_md] + radios + narr_radios + [comment]
|
| 262 |
|
| 263 |
def start(name):
|
| 264 |
name = (name or "").strip()
|
|
|
|
| 274 |
skip_btn.click(lambda i, n: show(i + 1, n), [idx_state, name_state], outputs)
|
| 275 |
|
| 276 |
def submit(i, name, *vals):
|
| 277 |
+
answers, narr_ans, cmt = vals[:len(DIMENSIONS)], vals[len(DIMENSIONS):-1], vals[-1]
|
|
|
|
|
|
|
| 278 |
it = ITEMS[i]
|
| 279 |
+
ns = narrations_of(it)
|
| 280 |
+
need = max(1, len(ns))
|
| 281 |
+
if any(a is None for a in answers) or any(a is None for a in narr_ans[:need]):
|
| 282 |
+
raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions and all {need} narration question(s) before submitting.")
|
| 283 |
row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name,
|
| 284 |
"video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"],
|
| 285 |
"comment": (cmt or "").strip()}
|
| 286 |
row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)})
|
| 287 |
+
row["narration_ids"] = ";".join(str(e.get("id")) for e in ns) or "(none)"
|
| 288 |
+
row.update({f"narration_{k + 1}": (narr_ans[k] if k < need else "") for k in range(NARR_MAX)})
|
| 289 |
append_row(row)
|
| 290 |
if i + 1 >= N:
|
| 291 |
return show(i, name) + [f"**Saved.** That was the last item β all {N} done. Thank you!"]
|
| 292 |
return show(i + 1, name) + [f"Saved `{it['object_id']}` β {RESP_CSV.name}"]
|
| 293 |
|
| 294 |
+
submit_btn.click(submit, [idx_state, name_state] + radios + narr_radios + [comment], outputs + [msg])
|
| 295 |
|
| 296 |
if __name__ == "__main__":
|
| 297 |
ap = argparse.ArgumentParser()
|