Card: import prompt constants from primo_prompts / frame helper from primo_video_utils
Browse files
README.md
CHANGED
|
@@ -96,89 +96,31 @@ The prompt template is equally load-bearing: the answer extractor is a regex ove
|
|
| 96 |
|
| 97 |
This example is transcribed from `src/eval/eval_interleave.py`, the harness that produced the published numbers.
|
| 98 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 99 |
```python
|
| 100 |
-
import cv2
|
| 101 |
import torch
|
| 102 |
-
from PIL import Image
|
| 103 |
from transformers import AutoProcessor, AutoTokenizer
|
| 104 |
from vllm import LLM, SamplingParams
|
| 105 |
from qwen_vl_utils import process_vision_info
|
| 106 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 107 |
MODEL_PATH = "models/PRIMO-R1-7B" # or "LeonOverload/PRIMO-R1-7B"
|
| 108 |
video_path = "path/to/your/episode.mp4"
|
| 109 |
question = "What is the completion percentage of the task in the video?"
|
| 110 |
problem_type = "regression"
|
| 111 |
|
| 112 |
-
|
| 113 |
-
|
| 114 |
-
|
| 115 |
-
"The reasoning process is enclosed within <think> </think> tags, and the final answer is within <answer> </answer> tags. "
|
| 116 |
-
"The <think> block must contain three ordered subsections: <planning>, <observation>, and <reasoning>. "
|
| 117 |
-
"The <answer> block must contain only the final output required by the question type and no other commentary."
|
| 118 |
-
)
|
| 119 |
-
|
| 120 |
-
QUESTION_TEMPLATE = (
|
| 121 |
-
"QUESTION:\n{Question}\n\n"
|
| 122 |
-
"QUESTION TYPE:\n{question_type}\n\n"
|
| 123 |
-
"Analyze the provided visual data and reason about the ongoing task.\n\n"
|
| 124 |
-
"Please think about this question as if you were a human pondering deeply. "
|
| 125 |
-
"Provide your detailed reasoning between the <think> and </think> tags, following the subsections <planning>, <observation>, and <reasoning>. "
|
| 126 |
-
"Then give your final answer between the <answer> and </answer> tags.\n\n"
|
| 127 |
-
"Below is the required template:\n\n"
|
| 128 |
-
"<think>\n"
|
| 129 |
-
"<planning>\n"
|
| 130 |
-
"Identify the high-level goal of the agent, what is the initial state? What does successful completion look like?\n"
|
| 131 |
-
"Break down the high-level goal into a logical sequence of canonical steps. This serves as your mental plan for interpreting the task.\n"
|
| 132 |
-
"Use this plan to interpret actions, map observed behaviors to steps, assess progress, detect anomalies, and predict what happens next.\n"
|
| 133 |
-
"</planning>\n"
|
| 134 |
-
"<observation>\n"
|
| 135 |
-
"View the video as a temporal sequence of actions contributing to the procedure.\n"
|
| 136 |
-
"Objectively describe what is occurring in the current moment, noting evidence of progress or state changes.\n"
|
| 137 |
-
"Identify fine-grained actions and explain how they move the task forward.\n"
|
| 138 |
-
"List relevant objects, tools, and environmental context, emphasizing functional states and transformations.\n"
|
| 139 |
-
"Note cues—repetition, transitions, or completion indicators—that situate the action in the procedural script.\n"
|
| 140 |
-
"</observation>\n"
|
| 141 |
-
"<reasoning>\n"
|
| 142 |
-
"Think through the question as a human would, Engage in an internal dialogue using expressions such as 'let me think', 'wait', 'hmm', 'oh, I see', 'let's break it down', etc.\n"
|
| 143 |
-
"Connect observations to the procedural plan to determine which step is being executed, progress, correctness, or anomalies.\n"
|
| 144 |
-
"Reflect on assumptions, verify interpretations, and, if appropriate, predict the agent's next likely action.\n"
|
| 145 |
-
"Synthesize understanding of what the agent is doing, how it fits into the broader task, and whether the process seems successful.\n"
|
| 146 |
-
"You are encouraged to include self-reflection or verification in your reasoning process.\n"
|
| 147 |
-
"</reasoning>\n"
|
| 148 |
-
"</think>\n"
|
| 149 |
-
"<answer>\n"
|
| 150 |
-
"[Final answer here — strictly follow the `{question_type}` output format and include no extra commentary.]\n"
|
| 151 |
-
"</answer>"
|
| 152 |
-
)
|
| 153 |
-
|
| 154 |
-
TYPE_TEMPLATE = {
|
| 155 |
-
"multiple choice": " Please provide only the single option letter (e.g., A, B, C, D, etc.) within the <answer> </answer> tags.",
|
| 156 |
-
"numerical": " Please provide the numerical value (e.g., 42 or 3.14) within the <answer> </answer> tags.",
|
| 157 |
-
"OCR": " Please transcribe text from the image/video clearly and provide your text answer within the <answer> </answer> tags.",
|
| 158 |
-
"free-form": " Please provide your text answer within the <answer> </answer> tags.",
|
| 159 |
-
"regression": " Please provide the numerical value (e.g., 42 or 3.14) within the <answer> </answer> tags.",
|
| 160 |
-
"boolean": " Please provide only 'Yes' or 'No' as your answer within the <answer> </answer> tags.",
|
| 161 |
-
}
|
| 162 |
-
|
| 163 |
-
|
| 164 |
-
def extract_first_and_last_frame(path):
|
| 165 |
-
"""Initial state = first frame, current state = last decodable frame."""
|
| 166 |
-
cap = cv2.VideoCapture(path)
|
| 167 |
-
ok, first = cap.read()
|
| 168 |
-
if not ok:
|
| 169 |
-
raise RuntimeError(f"cannot read {path}")
|
| 170 |
-
total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
|
| 171 |
-
cap.set(cv2.CAP_PROP_POS_FRAMES, max(0, total - 1))
|
| 172 |
-
ok, last = cap.read()
|
| 173 |
-
if not ok: # some containers mis-report the count
|
| 174 |
-
cap.set(cv2.CAP_PROP_POS_FRAMES, max(0, total - 2))
|
| 175 |
-
ok, last = cap.read()
|
| 176 |
-
cap.release()
|
| 177 |
-
to_pil = lambda f: Image.fromarray(cv2.cvtColor(f, cv2.COLOR_BGR2RGB))
|
| 178 |
-
return to_pil(first), to_pil(last if ok else first)
|
| 179 |
-
|
| 180 |
-
|
| 181 |
-
init_img, current_img = extract_first_and_last_frame(video_path)
|
| 182 |
|
| 183 |
messages = [
|
| 184 |
{"role": "system", "content": [{"type": "text", "text": SYSTEM_PROMPT}]},
|
|
@@ -188,12 +130,8 @@ messages = [
|
|
| 188 |
{"type": "image", "image": init_img}, # 1. initial state
|
| 189 |
{"type": "video", "video": video_path, "nframes": 22}, # 2. the clip
|
| 190 |
{"type": "image", "image": current_img}, # 3. current state
|
| 191 |
-
|
| 192 |
-
|
| 193 |
-
"text": QUESTION_TEMPLATE.format(
|
| 194 |
-
Question=question, question_type=problem_type
|
| 195 |
-
) + TYPE_TEMPLATE[problem_type],
|
| 196 |
-
},
|
| 197 |
],
|
| 198 |
},
|
| 199 |
]
|
|
@@ -250,11 +188,25 @@ For `regression` and `numerical` questions the answer is a **progress percentage
|
|
| 250 |
|
| 251 |
### Question types
|
| 252 |
|
| 253 |
-
`regression` and `numerical` (progress estimation), `multiple choice`, `boolean` (failure detection), `free-form`, `OCR`.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 254 |
|
| 255 |
### A note on frame counts
|
| 256 |
|
| 257 |
-
The published results were produced with the per-video frame cap at 22, so the launcher's `--nframes 32` effectively sampled 22. `nframes=22` above reproduces that. Training capped videos at 16 frames; eval samples more at higher resolution.
|
| 258 |
|
| 259 |
## Performance
|
| 260 |
|
|
|
|
| 96 |
|
| 97 |
This example is transcribed from `src/eval/eval_interleave.py`, the harness that produced the published numbers.
|
| 98 |
|
| 99 |
+
The prompts and the frame extraction are **imported, not pasted**. Both live in the repo cloned during setup: `src/primo_prompts.py` holds every prompt constant and `src/primo_video_utils.py` holds the frame helpers. They are the same objects the eval harness uses, so an example that imports them cannot drift out of sync with the checkpoint. `setup.sh` puts `src/` on `PYTHONPATH`; from elsewhere, add it yourself:
|
| 100 |
+
|
| 101 |
+
```python
|
| 102 |
+
import sys
|
| 103 |
+
sys.path.insert(0, "/path/to/PRIMO-R1/src")
|
| 104 |
+
```
|
| 105 |
+
|
| 106 |
```python
|
|
|
|
| 107 |
import torch
|
|
|
|
| 108 |
from transformers import AutoProcessor, AutoTokenizer
|
| 109 |
from vllm import LLM, SamplingParams
|
| 110 |
from qwen_vl_utils import process_vision_info
|
| 111 |
|
| 112 |
+
# The single source of truth for the prompt format and the anchor frames.
|
| 113 |
+
from primo_prompts import SYSTEM_PROMPT, build_question
|
| 114 |
+
from primo_video_utils import extract_frames_on_demand
|
| 115 |
+
|
| 116 |
MODEL_PATH = "models/PRIMO-R1-7B" # or "LeonOverload/PRIMO-R1-7B"
|
| 117 |
video_path = "path/to/your/episode.mp4"
|
| 118 |
question = "What is the completion percentage of the task in the video?"
|
| 119 |
problem_type = "regression"
|
| 120 |
|
| 121 |
+
# (initial state, current state) as PIL images. LRU-cached, so calling this
|
| 122 |
+
# again for the same video is free.
|
| 123 |
+
init_img, current_img = extract_frames_on_demand(video_path)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 124 |
|
| 125 |
messages = [
|
| 126 |
{"role": "system", "content": [{"type": "text", "text": SYSTEM_PROMPT}]},
|
|
|
|
| 130 |
{"type": "image", "image": init_img}, # 1. initial state
|
| 131 |
{"type": "video", "video": video_path, "nframes": 22}, # 2. the clip
|
| 132 |
{"type": "image", "image": current_img}, # 3. current state
|
| 133 |
+
# QUESTION_TEMPLATE.format(...) + TYPE_TEMPLATE[problem_type]
|
| 134 |
+
{"type": "text", "text": build_question(question, problem_type)},
|
|
|
|
|
|
|
|
|
|
|
|
|
| 135 |
],
|
| 136 |
},
|
| 137 |
]
|
|
|
|
| 188 |
|
| 189 |
### Question types
|
| 190 |
|
| 191 |
+
`regression` and `numerical` (progress estimation), `multiple choice`, `boolean` (failure detection), `free-form`, `OCR`. `build_question` puts the type in both places — inside `QUESTION_TEMPLATE` and as the `TYPE_TEMPLATE` hint appended after it — because the model was trained with both present.
|
| 192 |
+
|
| 193 |
+
### The prompt constants
|
| 194 |
+
|
| 195 |
+
If you need to inspect or extend them rather than just call `build_question`:
|
| 196 |
+
|
| 197 |
+
```python
|
| 198 |
+
from primo_prompts import SYSTEM_PROMPT, QUESTION_TEMPLATE, TYPE_TEMPLATE
|
| 199 |
+
|
| 200 |
+
print(QUESTION_TEMPLATE.format(Question="...", question_type="regression"))
|
| 201 |
+
print(sorted(TYPE_TEMPLATE))
|
| 202 |
+
# ['OCR', 'boolean', 'free-form', 'multiple choice', 'numerical', 'regression']
|
| 203 |
+
```
|
| 204 |
+
|
| 205 |
+
`primo_prompts.py` also carries the baseline and ablation prompts under `*_BASELINE` / `*_PARSER_ONLY` / `*_QUESTION_ONLY` names. Those are for reproducing the paper's comparison rows, **not** for this checkpoint — it was trained on `QUESTION_TEMPLATE` and expects it.
|
| 206 |
|
| 207 |
### A note on frame counts
|
| 208 |
|
| 209 |
+
The published results were produced with the per-video frame cap at 22, so the launcher's `--nframes 32` effectively sampled 22. `nframes=22` above reproduces that; `primo_video_utils.choose_nframes` is what applies the cap in the harness (`MAX_NFRAMES`, overridable via `INTERLEAVE_MAX_NFRAMES`). Training capped videos at 16 frames; eval samples more at higher resolution.
|
| 210 |
|
| 211 |
## Performance
|
| 212 |
|