Download inference.py from tchbcb/samai-2b: direct link, hf CLI and curl.
- Browser
- Download file 2.84 kB
-
https://huggingface.co/tchbcb/samai-2b/resolve/main/inference.py
- Command line
-
hf download hf://tchbcb/samai-2b/inference.py
-
curl -L -o inference.py https://huggingface.co/tchbcb/samai-2b/resolve/main/inference.py
2.84 kB
| # -*- coding: utf-8 -*- | |
| """inference.py — samai-pnet-dmoe-2b 拉取即跑推理示例 | |
| 用法: | |
| python inference.py # 默认 HuggingFace 仓 | |
| python inference.py --repo /path/to/local # 本地目录 | |
| python inference.py --lru 4 # 启用 SSD-LRU 专家缓存模拟 (每层容量 4) | |
| 依赖: torch>=2.4, transformers>=5.0 (trust_remote_code 加载仓内 modeling_samai_pnet.py) | |
| """ | |
| import argparse | |
| import torch | |
| from transformers import AutoModelForCausalLM, AutoTokenizer | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--repo", default="tchbcb/samai-pnet-dmoe-2b") | |
| ap.add_argument("--lru", type=int, default=0, | |
| help=">0 启用 SSD-LRU 专家缓存 (每层容量; 需仓内 modeling 模块)") | |
| ap.add_argument("--max-new-tokens", type=int, default=256) | |
| args = ap.parse_args() | |
| tok = AutoTokenizer.from_pretrained(args.repo, trust_remote_code=True) | |
| model = AutoModelForCausalLM.from_pretrained(args.repo, trust_remote_code=True, | |
| dtype="auto").eval() | |
| device = "cuda" if torch.cuda.is_available() else "cpu" | |
| model.to(device) | |
| if args.lru: | |
| # SSDLRUExpertManager: 专家驻留 CPU(模拟 SSD), 路由激活时搬入, LRU 驱逐 | |
| from modeling_samai_pnet import SSDLRUExpertManager # remote_code 模块 | |
| mgr = SSDLRUExpertManager.install(model, capacity_per_layer=args.lru) | |
| print(f"[ssd-lru] capacity/layer={args.lru} (storage=cpu)") | |
| messages = [{"role": "user", "content": "一个长方形长 12 米宽 8 米,面积是多少?请简要推理。"}] | |
| enc = tok.apply_chat_template(messages, add_generation_prompt=True, | |
| tokenize=True, return_tensors="pt", | |
| return_dict=True).to(device) | |
| with torch.no_grad(): | |
| out = model.generate(**enc, max_new_tokens=args.max_new_tokens, | |
| do_sample=False, pad_token_id=tok.pad_token_id or 1) | |
| text = tok.decode(out[0][enc["input_ids"].shape[1]:], skip_special_tokens=True) | |
| print("=== 回答 ===") | |
| print(text.strip()) | |
| # Ponder 诊断: 每次 forward 追加一条 {mode, executed, steps_mean, ...} | |
| decode_entries = [e for e in model._ponder_log if e.get("mode") == "decode"] | |
| if decode_entries: | |
| steps = sum(e["steps_mean"] for e in decode_entries) / len(decode_entries) | |
| print(f"\n=== Ponder 诊断 ===\n本次生成平均思考步数: {steps:.2f} " | |
| f"(max_ponder_steps={model.config.max_ponder_steps})") | |
| if args.lru: | |
| s = mgr.summary() | |
| print(f"=== SSD-LRU ===\nhit={s['hit']} miss={s['miss']} " | |
| f"evict={s['evict']} hit_rate={s['hit_rate']:.3f}") | |
| mgr.remove() | |
| if __name__ == "__main__": | |
| main() | |