File size: 2,242 Bytes
2ee8896
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
# Processor probe: does the eval-path processor call actually embed the
# image? (NO model weights needed - processor only. Runs in seconds.)
import os, sys
os.environ["HF_HUB_DISABLE_XET"] = "1"
sys.path.insert(0, "/content/qwenjev")
sys.path.insert(0, "/content/qwenjev/train")
from pathlib import Path

from transformers import AutoProcessor

proc = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-2B-Instruct")

# two DIFFERENT 32x32 frames (drastic difference, like the 109-cell change)
gA = [[(r * 32 + c) % 16 for c in range(32)] for r in range(32)]
gB = [[((r * 32 + c) * 7 + 3) % 16 for c in range(32)] for r in range(32)]

from qwenjev.vision import render_grid_pil
imgA, imgB = render_grid_pil(gA), render_grid_pil(gB)
print("pil images:", imgA.size, imgB.size, imgA.mode)

prompt = "<|im_start|>user\nSTATE:\ntest state\n\nQUESTION:\nWhich candidate?<|im_end|>\n<|im_start|>assistant\n"

texts = [
    proc.apply_chat_template(
        [{"role": "user", "content": [{"type": "image"},
                                      {"type": "text", "text": l}]}],
        add_generation_prompt=True, tokenize=False)
    for l in ("CANDIDATE: ACTION1", "CANDIDATE: ACTION6")
]
print("--- chat template text[0] head ---")
print(texts[0][:220].replace("\n", "\\n"))

# exactly the eval-path call: images kwarg, padding, pt
batch = proc(text=texts, images=[imgA, imgB], padding=True, return_tensors="pt")
print("--- batch keys ---")
print(sorted(batch.keys()))
if "pixel_values" in batch:
    pv = batch["pixel_values"]
    print("pixel_values:", tuple(pv.shape), pv.dtype,
          "mean=", float(pv.float().mean()), "std=", float(pv.float().std()))
    print("image_grid_thw:", batch.get("image_grid_thw"))
    # vision token count inside input_ids (Qwen image pad id 151655)
    import torch
    n_vis = int((batch["input_ids"] == 151655).sum())
    print("vision pad tokens in input_ids:", n_vis)
else:
    print("!!! NO pixel_values: the eval path is TEXT-ONLY — BUG CONFIRMED")
# text-only comparison: same call WITHOUT images
batch2 = proc(text=[t.replace("[ vision content omitted ]", "") for t in texts],
              padding=True, return_tensors="pt")
print("--- no-image batch keys ---", sorted(batch2.keys()))
print("PROBE_DONE")