onw 0.3: --context at load time, top_k / penalties, tray app + setup scripts, --compact / --prune
d40db99 verified Download test_context.py from ryugyosoft/onw: direct link, hf CLI and curl.
- Browser
- Download file 1.39 kB
-
https://huggingface.co/ryugyosoft/onw/resolve/main/test_context.py
- Command line
-
hf download hf://ryugyosoft/onw/test_context.py
-
curl -L -o test_context.py https://huggingface.co/ryugyosoft/onw/resolve/main/test_context.py
1.39 kB
| """Context length set at load time (reshape): same answer on a short prompt, and a prompt longer than the | |
| converted 1024 tokens works. usage: python test_context.py MODEL_DIR CONTEXT [DEVICE]""" | |
| import sys | |
| from onw.chat import ChatEngine | |
| TEXT = ("東京は日本の首都であり、政治・経済・文化の中心地である。江戸時代には徳川幕府が置かれ、1868年の明治維新で東京と" | |
| "改称された。現在は約1400万人が暮らす世界有数の大都市で、多くの企業や大学が集まっている。") | |
| def run(e, q, n=60): | |
| out = [d for d in e.stream_chat([{"role": "user", "content": q}], n)] | |
| return "".join(x for x in out if isinstance(x, str)), out[-1] | |
| def main(): | |
| e = ChatEngine(sys.argv[1], sys.argv[3] if len(sys.argv) > 3 else None, context=int(sys.argv[2])) | |
| print("context:", e.max_context) | |
| a, st = run(e, "日本の首都はどこですか?一文で。") | |
| print("short:", repr(a)) | |
| long_q = "次の文章は何回同じ内容を繰り返していますか?最後に一言で要約してください。\n\n" + "\n".join( | |
| f"{i + 1}. {TEXT}" for i in range(22)) | |
| e.checkpoint = None | |
| b, st = run(e, long_q, 80) | |
| print(f"long prompt {st['prompt_tokens']} tokens, prefill {st['prefill_ms']:.0f} ms, {st['decode_tok_s']:.1f} tok/s:", repr(b)) | |
| if __name__ == "__main__": | |
| main() | |