| |
| |
| import os, sys |
| os.environ["HF_HUB_DISABLE_XET"] = "1" |
| sys.path.insert(0, "/content/qwenjev") |
| sys.path.insert(0, "/content/qwenjev/train") |
| from pathlib import Path |
|
|
| from transformers import AutoProcessor |
|
|
| proc = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-2B-Instruct") |
|
|
| |
| gA = [[(r * 32 + c) % 16 for c in range(32)] for r in range(32)] |
| gB = [[((r * 32 + c) * 7 + 3) % 16 for c in range(32)] for r in range(32)] |
|
|
| from qwenjev.vision import render_grid_pil |
| imgA, imgB = render_grid_pil(gA), render_grid_pil(gB) |
| print("pil images:", imgA.size, imgB.size, imgA.mode) |
|
|
| prompt = "<|im_start|>user\nSTATE:\ntest state\n\nQUESTION:\nWhich candidate?<|im_end|>\n<|im_start|>assistant\n" |
|
|
| texts = [ |
| proc.apply_chat_template( |
| [{"role": "user", "content": [{"type": "image"}, |
| {"type": "text", "text": l}]}], |
| add_generation_prompt=True, tokenize=False) |
| for l in ("CANDIDATE: ACTION1", "CANDIDATE: ACTION6") |
| ] |
| print("--- chat template text[0] head ---") |
| print(texts[0][:220].replace("\n", "\\n")) |
|
|
| |
| batch = proc(text=texts, images=[imgA, imgB], padding=True, return_tensors="pt") |
| print("--- batch keys ---") |
| print(sorted(batch.keys())) |
| if "pixel_values" in batch: |
| pv = batch["pixel_values"] |
| print("pixel_values:", tuple(pv.shape), pv.dtype, |
| "mean=", float(pv.float().mean()), "std=", float(pv.float().std())) |
| print("image_grid_thw:", batch.get("image_grid_thw")) |
| |
| import torch |
| n_vis = int((batch["input_ids"] == 151655).sum()) |
| print("vision pad tokens in input_ids:", n_vis) |
| else: |
| print("!!! NO pixel_values: the eval path is TEXT-ONLY — BUG CONFIRMED") |
| |
| batch2 = proc(text=[t.replace("[ vision content omitted ]", "") for t in texts], |
| padding=True, return_tensors="pt") |
| print("--- no-image batch keys ---", sorted(batch2.keys())) |
| print("PROBE_DONE") |
|
|