onw / test_image.py
ryugyosoft's picture
onw 0.2: renamed from npue; onw command; Qwen3.5 (dense) / Gemma 4 / vision support; LM head segment
49c3379 verified
Raw History Blame Contribute Delete
1.54 kB
"""Image chat on an onw model: turn 1 with an image, turn 2 reusing the kept state vs recomputed.
usage: python test_image.py MODEL_DIR IMAGE [DEVICE]"""
import sys, time
from PIL import Image
from onw.chat import ChatEngine
def run(e, msgs, n=60):
parts, st = [], None
for d in e.stream_chat([dict(m) for m in msgs], n):
if isinstance(d, dict):
st = d
else:
parts.append(d)
return "".join(parts), st
def main():
e = ChatEngine(sys.argv[1], sys.argv[3] if len(sys.argv) > 3 else None)
img = Image.open(sys.argv[2])
t1 = [{"role": "user", "content": [{"type": "image", "image": img}, {"type": "text", "text": "この画像に何が写っていますか?一文で。"}]}]
a1, s1 = run(e, t1)
print(f"turn 1: {a1!r}\n prompt {s1['prompt_tokens']} tok, vision {s1['vision_ms']:.0f} ms, prefill {s1['prefill_ms']:.0f} ms, "
f"decode {s1['decode_tok_s']:.1f} tok/s")
t2 = t1 + [{"role": "assistant", "content": a1}, {"role": "user", "content": "その動物の毛の色は?一言で。"}]
t0 = time.time()
a2, s2 = run(e, t2)
print(f"turn 2 (reuse): {a2!r}\n prompt {s2['prompt_tokens']} tok, reused {s2['cached_tokens']}, vision {s2['vision_ms']:.0f} ms, "
f"prefill {s2['prefill_ms']:.0f} ms, total {time.time()-t0:.1f}s")
e.checkpoint = None
t0 = time.time()
a3, s3 = run(e, t2)
print(f"turn 2 (scratch): {a3!r}\n total {time.time()-t0:.1f}s; identical={a2 == a3}")
if __name__ == "__main__":
main()