Text Generation
Transformers
Safetensors
mistral3
image-text-to-text
decision-model
typed-decisions
jev
jevbench
calibration
decode-free
multilingual
vision-language
conversational
Instructions to use StandardThinking/StandardOne-3B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use StandardThinking/StandardOne-3B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="StandardThinking/StandardOne-3B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("StandardThinking/StandardOne-3B") model = AutoModelForMultimodalLM.from_pretrained("StandardThinking/StandardOne-3B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use StandardThinking/StandardOne-3B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "StandardThinking/StandardOne-3B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/StandardThinking/StandardOne-3B
- SGLang
How to use StandardThinking/StandardOne-3B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-3B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-3B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use StandardThinking/StandardOne-3B with Docker Model Runner:
docker model run hf.co/StandardThinking/StandardOne-3B
Download server/benchmarks/run_matrix.py from StandardThinking/StandardOne-3B: direct link, hf CLI and curl.
- Browser
- Download file 9.36 kB
-
https://huggingface.co/StandardThinking/StandardOne-3B/resolve/main/server/benchmarks/run_matrix.py
- Command line
-
hf download hf://StandardThinking/StandardOne-3B/server/benchmarks/run_matrix.py
-
curl -L -o run_matrix.py https://huggingface.co/StandardThinking/StandardOne-3B/resolve/main/server/benchmarks/run_matrix.py
9.36 kB
| #!/usr/bin/env python3 | |
| """Run one pinned model at a time on one H200. Default is a command preview.""" | |
| import argparse | |
| import json | |
| import os | |
| import shlex | |
| import signal | |
| import socket | |
| import subprocess | |
| import sys | |
| import time | |
| from pathlib import Path | |
| ROOT = Path(__file__).resolve().parents[1] | |
| RUNTIME = ROOT / "benchmarks/h200/runtime.py" | |
| def commands(args, manifest, profile): | |
| root = args.output.resolve() / profile | |
| engine = [ | |
| sys.executable, | |
| str(RUNTIME), | |
| "launch", | |
| profile, | |
| "--engine-python", | |
| # Resolving a venv Python symlink loses pyvenv.cfg and its installed engine. | |
| str(args.engine_python.absolute()), | |
| "--gpu", | |
| args.gpu, | |
| "--output", | |
| str(root / "engine"), | |
| "--timeout", | |
| str(args.startup_timeout), | |
| "--execute", | |
| ] | |
| if args.probe_vision: | |
| engine.append("--probe-vision") | |
| adapter = [ | |
| sys.executable, | |
| "-m", | |
| "jev_adapter", | |
| "--engine-url", | |
| "http://127.0.0.1:30000", | |
| "--model", | |
| "decision-model", | |
| "--port", | |
| "30120", | |
| "--max-concurrency", | |
| "32", | |
| ] | |
| model = manifest["models"][profile] | |
| if model.get("tokenization") == "native_official_text": | |
| adapter += [ | |
| "--tokenizer-model", | |
| model["repo_id"], | |
| "--tokenizer-revision", | |
| model["revision"], | |
| ] | |
| suites = [ | |
| "decision-v7", | |
| "transfer-v4", | |
| "transfer-v9", | |
| "jevbench-original", | |
| "jevbench-easy", | |
| "jevbench-hard", | |
| ] | |
| if args.suite: | |
| suites = args.suite | |
| if args.include_external: | |
| suites += [ | |
| suite for suite in ["semif-v1", "scienthoon-v1"] if suite not in suites | |
| ] | |
| evaluations = [] | |
| for concurrency in args.concurrency: | |
| run = [ | |
| sys.executable, | |
| "-m", | |
| "jev_adapter.benchmarks.run", | |
| "--output", | |
| str(root / f"c{concurrency}"), | |
| "--engine-manifest", | |
| str(root / "engine/launch.json"), | |
| "--concurrency", | |
| str(concurrency), | |
| "--warmup", | |
| str(args.warmup), | |
| "--repeats", | |
| str(args.repeats), | |
| ] | |
| for suite in suites: | |
| split = "public" if suite.startswith("jevbench-") else "development" | |
| run += [ | |
| "--data", | |
| str(args.data_root.resolve() / suite / f"{split}.jsonl"), | |
| ] | |
| if args.limit is not None: | |
| run += ["--limit", str(args.limit)] | |
| evaluations.append(run) | |
| return { | |
| "profile": profile, | |
| "model": manifest["models"][profile], | |
| "engine": engine, | |
| "adapter": adapter, | |
| "evaluations": evaluations, | |
| } | |
| def stop_owned(process): | |
| if process is not None and process.poll() is None: | |
| # SIGINT can be inherited as ignored under nohup; the runtime handles TERM. | |
| os.killpg(process.pid, signal.SIGTERM) | |
| try: | |
| process.wait(timeout=40) | |
| except subprocess.TimeoutExpired as exc: | |
| raise RuntimeError( | |
| f"Owned process {process.pid} did not stop; check its log." | |
| ) from exc | |
| def wait_engine(process, ready_path, timeout): | |
| deadline = time.monotonic() + timeout | |
| while not ready_path.exists(): | |
| if process.poll() is not None: | |
| raise RuntimeError( | |
| "Engine launcher exited; see launcher.log and engine/engine.log" | |
| ) | |
| if time.monotonic() > deadline: | |
| raise TimeoutError("Engine startup deadline exceeded") | |
| time.sleep(1) | |
| def wait_adapter(process, timeout=60): | |
| import httpx | |
| deadline = time.monotonic() + timeout | |
| headers = {} | |
| if key := os.environ.get("JEV_API_KEY"): | |
| headers["Authorization"] = f"Bearer {key}" | |
| with httpx.Client(timeout=2) as client: | |
| while True: | |
| if process.poll() is not None: | |
| raise RuntimeError("Adapter startup failed; see adapter.log") | |
| try: | |
| response = client.get( | |
| "http://127.0.0.1:30120/v1/models", headers=headers | |
| ) | |
| if response.is_success: | |
| return | |
| except httpx.HTTPError: | |
| pass | |
| if time.monotonic() > deadline: | |
| raise TimeoutError("Adapter startup deadline exceeded") | |
| time.sleep(1) | |
| def main(): | |
| manifest = json.loads((RUNTIME.parent / "models.json").read_text()) | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--output", type=Path, required=True) | |
| parser.add_argument( | |
| "--engine-python", type=Path, default=ROOT / ".venv-sglang/bin/python" | |
| ) | |
| parser.add_argument("--data-root", type=Path, default=ROOT / "benchmarks/data") | |
| parser.add_argument("--profile", action="append", choices=list(manifest["models"])) | |
| parser.add_argument("--concurrency", type=int, action="append") | |
| parser.add_argument("--warmup", type=int, default=20) | |
| parser.add_argument("--repeats", type=int, default=1) | |
| parser.add_argument("--limit", type=int) | |
| parser.add_argument("--startup-timeout", type=int, default=3600) | |
| parser.add_argument("--gpu", default="0") | |
| parser.add_argument("--include-external", action="store_true") | |
| parser.add_argument( | |
| "--suite", | |
| action="append", | |
| choices=[ | |
| "decision-v7", | |
| "transfer-v4", | |
| "transfer-v9", | |
| "jevbench-original", | |
| "jevbench-easy", | |
| "jevbench-hard", | |
| "semif-v1", | |
| "scienthoon-v1", | |
| ], | |
| ) | |
| parser.add_argument( | |
| "--probe-vision", action="store_true", help="optional image readiness probe" | |
| ) | |
| parser.add_argument("--execute", action="store_true") | |
| args = parser.parse_args() | |
| args.concurrency = args.concurrency or [1] | |
| if ( | |
| any(c < 1 or c > 32 for c in args.concurrency) | |
| or len(set(args.concurrency)) != len(args.concurrency) | |
| or args.warmup < 0 | |
| or args.repeats < 1 | |
| or args.startup_timeout < 1 | |
| or (args.limit is not None and args.limit < 1) | |
| ): | |
| parser.error("invalid counts; concurrency must be unique values from 1 to 32") | |
| profiles = args.profile or manifest["default_matrix"] | |
| if len(set(profiles)) != len(profiles): | |
| parser.error("profiles must be unique") | |
| plans = [commands(args, manifest, profile) for profile in profiles] | |
| for plan in plans: | |
| print(f"\n{plan['profile']}") | |
| for command in [plan["engine"], plan["adapter"], *plan["evaluations"]]: | |
| print(shlex.join(command)) | |
| if not args.execute: | |
| print("\nPreview only. Add --execute on the prepared H200 server.") | |
| return | |
| if not args.engine_python.is_file(): | |
| parser.error("engine Python is missing; follow benchmarks/h200/README.md") | |
| for plan in plans: | |
| for command in plan["evaluations"]: | |
| for i, value in enumerate(command): | |
| if value == "--data" and not Path(command[i + 1]).is_file(): | |
| parser.error(f"missing prepared dataset: {command[i + 1]}") | |
| args.output.mkdir(parents=True, exist_ok=False) | |
| (args.output / "matrix.json").write_text(json.dumps(plans, indent=2) + "\n") | |
| for plan in plans: | |
| # Never attach to or stop unrelated services using the benchmark ports. | |
| for port in (30000, 30120): | |
| with socket.socket() as listener: | |
| # Match server bind semantics: ignore TIME_WAIT, reject live listeners. | |
| listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) | |
| listener.bind(("127.0.0.1", port)) | |
| root = args.output.resolve() / plan["profile"] | |
| root.mkdir() | |
| engine, adapter = None, None | |
| with ( | |
| (root / "launcher.log").open("x") as engine_log, | |
| (root / "adapter.log").open("x") as adapter_log, | |
| ): | |
| try: | |
| engine = subprocess.Popen( | |
| plan["engine"], | |
| cwd=ROOT, | |
| stdout=engine_log, | |
| stderr=subprocess.STDOUT, | |
| start_new_session=True, | |
| ) | |
| wait_engine( | |
| engine, root / "engine/ready.json", args.startup_timeout + 90 | |
| ) | |
| adapter = subprocess.Popen( | |
| plan["adapter"], | |
| cwd=ROOT, | |
| stdout=adapter_log, | |
| stderr=subprocess.STDOUT, | |
| start_new_session=True, | |
| ) | |
| wait_adapter(adapter) | |
| for command in plan["evaluations"]: | |
| subprocess.run(command, cwd=ROOT, check=True) | |
| finally: | |
| try: | |
| stop_owned(adapter) | |
| finally: | |
| stop_owned(engine) | |
| subprocess.run( | |
| [ | |
| sys.executable, | |
| "-m", | |
| "jev_adapter.benchmarks.compare", | |
| str(args.output.resolve()), | |
| "--output", | |
| str(args.output.resolve() / "comparison.csv"), | |
| ], | |
| cwd=ROOT, | |
| check=True, | |
| ) | |
| if __name__ == "__main__": | |
| main() | |