WitneyWW commited on
Commit
a4222ff
Β·
1 Parent(s): 4dc38f4

labelled VIDEO TIME clock + time axis

Browse files
Files changed (1) hide show
  1. app.py +55 -12
app.py CHANGED
@@ -28,11 +28,9 @@ DIMENSIONS = [
28
  ("localization", "2. Localization β€” do the red box / green mask cover the right object (not a neighbour, hand, or background)?"),
29
  ("state_history", "3. Location & state history β€” is the where / relation / timing (`now`, `history`) correct?"),
30
  ("moves", "4. Moves β€” are the recorded movement events (from β†’ to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"),
31
- ("narration", "5. Narrations β€” do the blue NARRATION texts (HD-EPIC annotator descriptions, with their start/end times) really describe this object, and are the times right?"),
32
- ("sound_speech", "6. Sound & speech β€” do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"),
33
- ("gaze", "7. Gaze β€” is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"),
34
  ]
35
- COLUMNS = ["timestamp", "annotator", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS] + ["comment"]
36
 
37
  # ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ----
38
  DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses
@@ -53,6 +51,42 @@ if DATASET_REPO and os.environ.get("HF_TOKEN"):
53
  allow_patterns=["*.csv"])
54
 
55
  ITEMS = json.load(open(MEDIA / "index.json"))
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
56
  N = len(ITEMS)
57
 
58
 
@@ -169,7 +203,8 @@ def show(idx, annotator):
169
  done = load_done(annotator) if annotator else set()
170
  status = "βœ… answered" if item_key(it) in done else "⬜ not answered"
171
  header = f"### {annotator or '?'} β€” item {idx + 1} / {N}   {status}   ({len(done)} / {N} done)"
172
- return [idx, header, str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)] + [None] * len(DIMENSIONS) + [""]
 
173
 
174
 
175
  def first_unanswered(annotator):
@@ -193,8 +228,9 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
193
  with gr.Column(visible=False) as main_col:
194
  header = gr.Markdown()
195
  gr.Markdown(
196
- "All times are **seconds in the full recording** (shown top-right of the video and on the timeline), "
197
- "not seconds since the clip started. "
 
198
  "**Clip overlays:** red box = object near each keyframe Β· red dot = where the wearer is looking (gaze) Β· "
199
  "yellow *GAZE ON OBJECT* = the memory claims a gaze dwell Β· **blue NARRATION** = HD-EPIC annotator text "
200
  "(not audio) with its [start-end] and START/END markers Β· **purple SPEECH** = words actually spoken in the "
@@ -212,6 +248,9 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
212
  xc_md = gr.Markdown()
213
  gr.Markdown("## Your judgement")
214
  radios = [gr.Radio(CHOICES, label=q) for _, q in DIMENSIONS]
 
 
 
215
  comment = gr.Textbox(label="Comment (optional)", lines=2)
216
  with gr.Row():
217
  prev_btn = gr.Button("β—€ Prev")
@@ -219,7 +258,7 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
219
  submit_btn = gr.Button("Submit & Next β–Ά", variant="primary")
220
  msg = gr.Markdown()
221
 
222
- outputs = [idx_state, header, video, gallery, ann_md, xc_md] + radios + [comment]
223
 
224
  def start(name):
225
  name = (name or "").strip()
@@ -235,20 +274,24 @@ with gr.Blocks(title="Object-centric annotation eval") as demo:
235
  skip_btn.click(lambda i, n: show(i + 1, n), [idx_state, name_state], outputs)
236
 
237
  def submit(i, name, *vals):
238
- *answers, cmt = vals
239
- if any(a is None for a in answers):
240
- raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions before submitting.")
241
  it = ITEMS[i]
 
 
 
 
242
  row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name,
243
  "video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"],
244
  "comment": (cmt or "").strip()}
245
  row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)})
 
 
246
  append_row(row)
247
  if i + 1 >= N:
248
  return show(i, name) + [f"**Saved.** That was the last item β€” all {N} done. Thank you!"]
249
  return show(i + 1, name) + [f"Saved `{it['object_id']}` β†’ {RESP_CSV.name}"]
250
 
251
- submit_btn.click(submit, [idx_state, name_state] + radios + [comment], outputs + [msg])
252
 
253
  if __name__ == "__main__":
254
  ap = argparse.ArgumentParser()
 
28
  ("localization", "2. Localization β€” do the red box / green mask cover the right object (not a neighbour, hand, or background)?"),
29
  ("state_history", "3. Location & state history β€” is the where / relation / timing (`now`, `history`) correct?"),
30
  ("moves", "4. Moves β€” are the recorded movement events (from β†’ to, time) correct? (If `moves` is empty: is it correct that the object did not move?)"),
31
+ ("sound_speech", "5. Sound & speech β€” do the green SOUND events and any purple SPEECH (ASR of the audio) really belong to this object?"),
32
+ ("gaze", "6. Gaze β€” is the `looked at` claim right, i.e. does the red gaze dot sit on this object during the yellow `dwell` intervals?"),
 
33
  ]
 
34
 
35
  # ---- optional persistence to a HF Dataset (used on HF Spaces, where local disk is ephemeral) ----
36
  DATASET_REPO = os.environ.get("RESPONSES_DATASET") # e.g. WitneyWW/object-memory-eval-responses
 
51
  allow_patterns=["*.csv"])
52
 
53
  ITEMS = json.load(open(MEDIA / "index.json"))
54
+ # Narrations are judged one by one. An object has between 0 and NARR_MAX of them, so the
55
+ # page holds NARR_MAX radio groups and shows only as many as the current object needs.
56
+ NARR_MAX = max(1, max(len(it["events"]["narration"]) for it in ITEMS))
57
+ COLUMNS = (["timestamp", "annotator", "video_id", "chunk", "object_id"] + [d[0] for d in DIMENSIONS]
58
+ + ["narration_ids"] + [f"narration_{k + 1}" for k in range(NARR_MAX)] + ["comment"])
59
+ NO_NARR_Q = ("No narration is linked to this object. Is that right β€” is there really no narration in this clip "
60
+ "about it? (All narrations of the clip are listed in the cross-check panel.)")
61
+
62
+ # A responses file written with a different column set (older version of the form) is kept
63
+ # under another name instead of being appended to with misaligned columns.
64
+ if RESP_CSV.exists():
65
+ with open(RESP_CSV, newline="") as _f:
66
+ _hdr = next(csv.reader(_f), None)
67
+ if _hdr != COLUMNS:
68
+ RESP_CSV.rename(RESP_DIR / f"responses_old_{datetime.now():%Y%m%d_%H%M%S}.csv")
69
+
70
+
71
+ def narrations_of(it):
72
+ return sorted(it["events"]["narration"], key=lambda e: e["t0"])
73
+
74
+
75
+ def narration_updates(it):
76
+ """gr.update for each of the NARR_MAX radio groups: label + visibility for this object."""
77
+ ns = narrations_of(it)
78
+ ups = []
79
+ for k in range(NARR_MAX):
80
+ if k < len(ns):
81
+ e = ns[k]
82
+ ups.append(gr.update(visible=True, value=None,
83
+ label=f"Narration {k + 1} of {len(ns)} β€” [{e['t0']:.1f}–{e['t1']:.1f}s] β€œ{e['label']}” β€” "
84
+ "is this about THIS object, and is the time right?"))
85
+ elif k == 0:
86
+ ups.append(gr.update(visible=True, value=None, label=NO_NARR_Q))
87
+ else:
88
+ ups.append(gr.update(visible=False, value=None))
89
+ return ups
90
  N = len(ITEMS)
91
 
92
 
 
203
  done = load_done(annotator) if annotator else set()
204
  status = "βœ… answered" if item_key(it) in done else "⬜ not answered"
205
  header = f"### {annotator or '?'} β€” item {idx + 1} / {N} &nbsp; {status} &nbsp; ({len(done)} / {N} done)"
206
+ return ([idx, header, str(MEDIA / it["clip"]), gallery_for(it), annotation_md(it), crosscheck_md(it)]
207
+ + [None] * len(DIMENSIONS) + narration_updates(it) + [""])
208
 
209
 
210
  def first_unanswered(annotator):
 
228
  with gr.Column(visible=False) as main_col:
229
  header = gr.Markdown()
230
  gr.Markdown(
231
+ "⏱ **All annotation times are seconds in the full recording.** The current time is the boxed "
232
+ "**VIDEO TIME** at the top-right of the video and the white playhead on the bottom time axis "
233
+ "(the player's own 0–30 s counter is NOT the annotation time). "
234
  "**Clip overlays:** red box = object near each keyframe Β· red dot = where the wearer is looking (gaze) Β· "
235
  "yellow *GAZE ON OBJECT* = the memory claims a gaze dwell Β· **blue NARRATION** = HD-EPIC annotator text "
236
  "(not audio) with its [start-end] and START/END markers Β· **purple SPEECH** = words actually spoken in the "
 
248
  xc_md = gr.Markdown()
249
  gr.Markdown("## Your judgement")
250
  radios = [gr.Radio(CHOICES, label=q) for _, q in DIMENSIONS]
251
+ gr.Markdown("### Narrations β€” judge each one separately\n"
252
+ "Blue NARRATION subtitles in the video; compare the [start–end] with the VIDEO TIME clock.")
253
+ narr_radios = [gr.Radio(CHOICES, label=f"Narration {k + 1}", visible=(k == 0)) for k in range(NARR_MAX)]
254
  comment = gr.Textbox(label="Comment (optional)", lines=2)
255
  with gr.Row():
256
  prev_btn = gr.Button("β—€ Prev")
 
258
  submit_btn = gr.Button("Submit & Next β–Ά", variant="primary")
259
  msg = gr.Markdown()
260
 
261
+ outputs = [idx_state, header, video, gallery, ann_md, xc_md] + radios + narr_radios + [comment]
262
 
263
  def start(name):
264
  name = (name or "").strip()
 
274
  skip_btn.click(lambda i, n: show(i + 1, n), [idx_state, name_state], outputs)
275
 
276
  def submit(i, name, *vals):
277
+ answers, narr_ans, cmt = vals[:len(DIMENSIONS)], vals[len(DIMENSIONS):-1], vals[-1]
 
 
278
  it = ITEMS[i]
279
+ ns = narrations_of(it)
280
+ need = max(1, len(ns))
281
+ if any(a is None for a in answers) or any(a is None for a in narr_ans[:need]):
282
+ raise gr.Error(f"Please answer all {len(DIMENSIONS)} questions and all {need} narration question(s) before submitting.")
283
  row = {"timestamp": datetime.now().isoformat(timespec="seconds"), "annotator": name,
284
  "video_id": it["video_id"], "chunk": it["chunk"], "object_id": it["object_id"],
285
  "comment": (cmt or "").strip()}
286
  row.update({d[0]: a for d, a in zip(DIMENSIONS, answers)})
287
+ row["narration_ids"] = ";".join(str(e.get("id")) for e in ns) or "(none)"
288
+ row.update({f"narration_{k + 1}": (narr_ans[k] if k < need else "") for k in range(NARR_MAX)})
289
  append_row(row)
290
  if i + 1 >= N:
291
  return show(i, name) + [f"**Saved.** That was the last item β€” all {N} done. Thank you!"]
292
  return show(i + 1, name) + [f"Saved `{it['object_id']}` β†’ {RESP_CSV.name}"]
293
 
294
+ submit_btn.click(submit, [idx_state, name_state] + radios + narr_radios + [comment], outputs + [msg])
295
 
296
  if __name__ == "__main__":
297
  ap = argparse.ArgumentParser()