qwenjev / scripts /proc_probe.py
tchbcb's picture
v0.4.9: RCA evidence archived + HANDOVER §0.4
2ee8896 verified
Raw
History Blame Contribute Delete
2.24 kB
# Processor probe: does the eval-path processor call actually embed the
# image? (NO model weights needed - processor only. Runs in seconds.)
import os, sys
os.environ["HF_HUB_DISABLE_XET"] = "1"
sys.path.insert(0, "/content/qwenjev")
sys.path.insert(0, "/content/qwenjev/train")
from pathlib import Path
from transformers import AutoProcessor
proc = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-2B-Instruct")
# two DIFFERENT 32x32 frames (drastic difference, like the 109-cell change)
gA = [[(r * 32 + c) % 16 for c in range(32)] for r in range(32)]
gB = [[((r * 32 + c) * 7 + 3) % 16 for c in range(32)] for r in range(32)]
from qwenjev.vision import render_grid_pil
imgA, imgB = render_grid_pil(gA), render_grid_pil(gB)
print("pil images:", imgA.size, imgB.size, imgA.mode)
prompt = "<|im_start|>user\nSTATE:\ntest state\n\nQUESTION:\nWhich candidate?<|im_end|>\n<|im_start|>assistant\n"
texts = [
proc.apply_chat_template(
[{"role": "user", "content": [{"type": "image"},
{"type": "text", "text": l}]}],
add_generation_prompt=True, tokenize=False)
for l in ("CANDIDATE: ACTION1", "CANDIDATE: ACTION6")
]
print("--- chat template text[0] head ---")
print(texts[0][:220].replace("\n", "\\n"))
# exactly the eval-path call: images kwarg, padding, pt
batch = proc(text=texts, images=[imgA, imgB], padding=True, return_tensors="pt")
print("--- batch keys ---")
print(sorted(batch.keys()))
if "pixel_values" in batch:
pv = batch["pixel_values"]
print("pixel_values:", tuple(pv.shape), pv.dtype,
"mean=", float(pv.float().mean()), "std=", float(pv.float().std()))
print("image_grid_thw:", batch.get("image_grid_thw"))
# vision token count inside input_ids (Qwen image pad id 151655)
import torch
n_vis = int((batch["input_ids"] == 151655).sum())
print("vision pad tokens in input_ids:", n_vis)
else:
print("!!! NO pixel_values: the eval path is TEXT-ONLY — BUG CONFIRMED")
# text-only comparison: same call WITHOUT images
batch2 = proc(text=[t.replace("[ vision content omitted ]", "") for t in texts],
padding=True, return_tensors="pt")
print("--- no-image batch keys ---", sorted(batch2.keys()))
print("PROBE_DONE")