Text Generation
Transformers
Safetensors
mistral3
image-text-to-text
decision-model
typed-decisions
jev
jevbench
calibration
decode-free
multilingual
vision-language
conversational
Instructions to use StandardThinking/StandardOne-3B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use StandardThinking/StandardOne-3B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="StandardThinking/StandardOne-3B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("StandardThinking/StandardOne-3B") model = AutoModelForMultimodalLM.from_pretrained("StandardThinking/StandardOne-3B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use StandardThinking/StandardOne-3B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "StandardThinking/StandardOne-3B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/StandardThinking/StandardOne-3B
- SGLang
How to use StandardThinking/StandardOne-3B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-3B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-3B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use StandardThinking/StandardOne-3B with Docker Model Runner:
docker model run hf.co/StandardThinking/StandardOne-3B
Download server/tests/test_main.py from StandardThinking/StandardOne-3B: direct link, hf CLI and curl.
- Browser
- Download file 11.5 kB
-
https://huggingface.co/StandardThinking/StandardOne-3B/resolve/main/server/tests/test_main.py
- Command line
-
hf download hf://StandardThinking/StandardOne-3B/server/tests/test_main.py
-
curl -L -o test_main.py https://huggingface.co/StandardThinking/StandardOne-3B/resolve/main/server/tests/test_main.py
11.5 kB
| """CLI wiring tests for --prompt-wording (argparse choices/env default, and | |
| plumbing into SGLangBackend + create_app). uvicorn.run is monkeypatched so | |
| main() never actually binds a socket; SGLangBackend/create_app/NativeTokenizer | |
| are monkeypatched to capture their constructor arguments instead of talking to | |
| a real engine or loading real tokenizer files (transformers is an optional, | |
| not-installed-here dependency for this default venv).""" | |
| import sys | |
| import pytest | |
| import jev_adapter.__main__ as main_module | |
| import jev_adapter.native_tokenizer as native_tokenizer_module | |
| import jev_adapter.server as server_module | |
| import jev_adapter.sglang as sglang_module | |
| class Capture: | |
| def __init__(self): | |
| self.backend_kwargs = None | |
| self.app_kwargs = None | |
| def fake_backend(self, *args, **kwargs): | |
| self.backend_kwargs = kwargs | |
| return object() | |
| def fake_create_app(self, backend, **kwargs): | |
| self.app_kwargs = kwargs | |
| return object() | |
| def capture(monkeypatch): | |
| capture = Capture() | |
| monkeypatch.setattr(sglang_module, "SGLangBackend", capture.fake_backend) | |
| monkeypatch.setattr(server_module, "create_app", capture.fake_create_app) | |
| monkeypatch.setattr("uvicorn.run", lambda app, **kwargs: None) | |
| return capture | |
| def run_main(monkeypatch, argv, env=None): | |
| monkeypatch.setattr(sys, "argv", ["jev-adapter", *argv]) | |
| for key in ( | |
| "JEV_PROMPT_WORDING", | |
| "JEV_DEFAULT_TEMPERATURE", | |
| "JEV_NATIVE_SYSTEM_PROMPT", | |
| ): | |
| monkeypatch.delenv(key, raising=False) | |
| for key, value in (env or {}).items(): | |
| monkeypatch.setenv(key, value) | |
| main_module.main() | |
| def test_prompt_wording_defaults_to_served(monkeypatch, capture): | |
| run_main(monkeypatch, ["--model", "decision-model"]) | |
| assert capture.backend_kwargs["prompt_wording"] == "served" | |
| assert capture.backend_kwargs["native_system_prompt"] is None | |
| assert capture.app_kwargs["prompt_wording"] == "served" | |
| def test_prompt_wording_flag_selects_native(monkeypatch, capture): | |
| run_main(monkeypatch, ["--model", "decision-model", "--prompt-wording", "native"]) | |
| assert capture.backend_kwargs["prompt_wording"] == "native" | |
| assert capture.app_kwargs["prompt_wording"] == "native" | |
| def test_prompt_wording_env_var_sets_the_default(monkeypatch, capture): | |
| run_main( | |
| monkeypatch, | |
| ["--model", "decision-model"], | |
| env={"JEV_PROMPT_WORDING": "native"}, | |
| ) | |
| assert capture.backend_kwargs["prompt_wording"] == "native" | |
| assert capture.app_kwargs["prompt_wording"] == "native" | |
| def test_explicit_flag_overrides_env_var(monkeypatch, capture): | |
| run_main( | |
| monkeypatch, | |
| ["--model", "decision-model", "--prompt-wording", "served"], | |
| env={"JEV_PROMPT_WORDING": "native"}, | |
| ) | |
| assert capture.backend_kwargs["prompt_wording"] == "served" | |
| def test_invalid_prompt_wording_choice_rejected(monkeypatch, capture): | |
| monkeypatch.setattr(sys, "argv", ["jev-adapter", "--model", "m", "--prompt-wording", "bogus"]) | |
| with pytest.raises(SystemExit): | |
| main_module.main() | |
| def test_native_wording_without_tokenizer_model_serves_text_only_and_warns( | |
| monkeypatch, capture, caplog | |
| ): | |
| with caplog.at_level("WARNING"): | |
| run_main(monkeypatch, ["--model", "decision-model", "--prompt-wording", "native"]) | |
| assert capture.backend_kwargs["native_system_prompt"] is None | |
| assert any("native" in record.message for record in caplog.records) | |
| def test_native_wording_with_tokenizer_model_extracts_the_system_prompt( | |
| monkeypatch, capture | |
| ): | |
| calls = [] | |
| class FakeNativeTokenizer: | |
| def from_pretrained(cls, model, revision): | |
| calls.append(("from_pretrained", model, revision)) | |
| return object() | |
| def native_default_system_prompt(cls, model, revision): | |
| calls.append(("native_default_system_prompt", model, revision)) | |
| return "Default system text." | |
| monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer) | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--prompt-wording", | |
| "native", | |
| "--tokenizer-model", | |
| "mistralai/Ministral-3-8B-Instruct-2512-BF16", | |
| "--tokenizer-revision", | |
| "f" * 40, | |
| ], | |
| ) | |
| assert calls == [ | |
| ("from_pretrained", "mistralai/Ministral-3-8B-Instruct-2512-BF16", "f" * 40), | |
| ( | |
| "native_default_system_prompt", | |
| "mistralai/Ministral-3-8B-Instruct-2512-BF16", | |
| "f" * 40, | |
| ), | |
| ] | |
| assert capture.backend_kwargs["native_system_prompt"] == "Default system text." | |
| def test_served_wording_with_tokenizer_model_never_extracts_a_system_prompt( | |
| monkeypatch, capture | |
| ): | |
| calls = [] | |
| class FakeNativeTokenizer: | |
| def from_pretrained(cls, model, revision): | |
| calls.append("from_pretrained") | |
| return object() | |
| def native_default_system_prompt(cls, model, revision): | |
| calls.append("native_default_system_prompt") | |
| return "should not be reached" | |
| monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer) | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--tokenizer-model", | |
| "org/model", | |
| "--tokenizer-revision", | |
| "a" * 40, | |
| ], | |
| ) | |
| assert calls == ["from_pretrained"] | |
| assert capture.backend_kwargs["prompt_wording"] == "served" | |
| assert capture.backend_kwargs["native_system_prompt"] is None | |
| def test_native_system_prompt_defaults_to_auto(monkeypatch, capture): | |
| calls = [] | |
| class FakeNativeTokenizer: | |
| def from_pretrained(cls, model, revision): | |
| return object() | |
| def native_default_system_prompt(cls, model, revision): | |
| calls.append((model, revision)) | |
| return "Default system text." | |
| monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer) | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--prompt-wording", | |
| "native", | |
| "--tokenizer-model", | |
| "org/model", | |
| "--tokenizer-revision", | |
| "a" * 40, | |
| ], | |
| ) | |
| assert calls == [("org/model", "a" * 40)] | |
| assert capture.backend_kwargs["native_system_prompt"] == "Default system text." | |
| def test_native_system_prompt_none_skips_extraction_and_warning( | |
| monkeypatch, capture, caplog | |
| ): | |
| calls = [] | |
| class FakeNativeTokenizer: | |
| def from_pretrained(cls, model, revision): | |
| return object() | |
| def native_default_system_prompt(cls, model, revision): | |
| calls.append((model, revision)) | |
| return "should not be reached" | |
| monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer) | |
| with caplog.at_level("WARNING"): | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--prompt-wording", | |
| "native", | |
| "--tokenizer-model", | |
| "org/model", | |
| "--tokenizer-revision", | |
| "a" * 40, | |
| "--native-system-prompt", | |
| "none", | |
| ], | |
| ) | |
| assert calls == [] | |
| assert capture.backend_kwargs["native_system_prompt"] is None | |
| assert not any("native" in record.message for record in caplog.records) | |
| def test_native_system_prompt_none_without_tokenizer_never_warns( | |
| monkeypatch, capture, caplog | |
| ): | |
| with caplog.at_level("WARNING"): | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--prompt-wording", | |
| "native", | |
| "--native-system-prompt", | |
| "none", | |
| ], | |
| ) | |
| assert capture.backend_kwargs["native_system_prompt"] is None | |
| assert caplog.records == [] | |
| def test_native_system_prompt_env_var_sets_the_default(monkeypatch, capture, caplog): | |
| with caplog.at_level("WARNING"): | |
| run_main( | |
| monkeypatch, | |
| ["--model", "decision-model", "--prompt-wording", "native"], | |
| env={"JEV_NATIVE_SYSTEM_PROMPT": "none"}, | |
| ) | |
| assert capture.backend_kwargs["native_system_prompt"] is None | |
| assert caplog.records == [] | |
| def test_native_system_prompt_explicit_flag_overrides_env_var(monkeypatch, capture): | |
| calls = [] | |
| class FakeNativeTokenizer: | |
| def from_pretrained(cls, model, revision): | |
| return object() | |
| def native_default_system_prompt(cls, model, revision): | |
| calls.append((model, revision)) | |
| return "Default system text." | |
| monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer) | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--prompt-wording", | |
| "native", | |
| "--tokenizer-model", | |
| "org/model", | |
| "--tokenizer-revision", | |
| "a" * 40, | |
| "--native-system-prompt", | |
| "auto", | |
| ], | |
| env={"JEV_NATIVE_SYSTEM_PROMPT": "none"}, | |
| ) | |
| assert calls == [("org/model", "a" * 40)] | |
| assert capture.backend_kwargs["native_system_prompt"] == "Default system text." | |
| def test_native_system_prompt_none_has_no_effect_on_served_wording( | |
| monkeypatch, capture | |
| ): | |
| calls = [] | |
| class FakeNativeTokenizer: | |
| def from_pretrained(cls, model, revision): | |
| return object() | |
| def native_default_system_prompt(cls, model, revision): | |
| calls.append((model, revision)) | |
| return "should not be reached" | |
| monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer) | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--tokenizer-model", | |
| "org/model", | |
| "--tokenizer-revision", | |
| "a" * 40, | |
| "--native-system-prompt", | |
| "auto", | |
| ], | |
| ) | |
| assert calls == [] | |
| assert capture.backend_kwargs["prompt_wording"] == "served" | |
| assert capture.backend_kwargs["native_system_prompt"] is None | |
| def test_invalid_native_system_prompt_choice_rejected(monkeypatch, capture): | |
| monkeypatch.setattr( | |
| sys, | |
| "argv", | |
| ["jev-adapter", "--model", "m", "--native-system-prompt", "bogus"], | |
| ) | |
| with pytest.raises(SystemExit): | |
| main_module.main() | |
| def test_native_system_prompt_startup_log_line(monkeypatch, capture, caplog): | |
| with caplog.at_level("INFO"): | |
| run_main( | |
| monkeypatch, | |
| [ | |
| "--model", | |
| "decision-model", | |
| "--prompt-wording", | |
| "native", | |
| "--native-system-prompt", | |
| "none", | |
| ], | |
| ) | |
| info_records = [r for r in caplog.records if r.levelname == "INFO"] | |
| assert len(info_records) == 1 | |
| message = info_records[0].getMessage() | |
| assert "prompt_wording=native" in message | |
| assert "native_system_prompt=none" in message | |