Text Generation
Transformers
Safetensors
mistral3
image-text-to-text
decision-model
typed-decisions
jev
jevbench
calibration
decode-free
multilingual
vision-language
conversational
Instructions to use StandardThinking/StandardOne-3B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use StandardThinking/StandardOne-3B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="StandardThinking/StandardOne-3B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("StandardThinking/StandardOne-3B") model = AutoModelForMultimodalLM.from_pretrained("StandardThinking/StandardOne-3B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use StandardThinking/StandardOne-3B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "StandardThinking/StandardOne-3B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/StandardThinking/StandardOne-3B
- SGLang
How to use StandardThinking/StandardOne-3B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-3B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-3B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-3B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use StandardThinking/StandardOne-3B with Docker Model Runner:
docker model run hf.co/StandardThinking/StandardOne-3B
Update weights
Browse files- MERGE_REPORT.json +12 -12
- README.md +13 -3
- SHA256SUMS +6 -6
- docs/BENCHMARKS.md +55 -10
- model-00001-of-00002.safetensors +1 -1
- model-00002-of-00002.safetensors +1 -1
- release-manifest.json +6 -4
MERGE_REPORT.json
CHANGED
|
@@ -1,15 +1,15 @@
|
|
| 1 |
{
|
| 2 |
"snapshot": "mistralai/Ministral-3-3B-Instruct-2512-BF16@b6d637bef2393152b3da2b2fde72eecdee30557e",
|
| 3 |
-
"adapter": "StandardOne-3B-LoRA (training run
|
| 4 |
-
"adapter_sha256": "
|
| 5 |
"changed_params": 182,
|
| 6 |
"unchanged_params": 276,
|
| 7 |
"changed_outside_lm_projections": [],
|
| 8 |
-
"max_abs_delta": 0.
|
| 9 |
"changed_sample": [
|
| 10 |
[
|
| 11 |
"model.language_model.layers.0.self_attn.q_proj.weight",
|
| 12 |
-
0.
|
| 13 |
],
|
| 14 |
[
|
| 15 |
"model.language_model.layers.0.self_attn.k_proj.weight",
|
|
@@ -17,19 +17,19 @@
|
|
| 17 |
],
|
| 18 |
[
|
| 19 |
"model.language_model.layers.0.self_attn.v_proj.weight",
|
| 20 |
-
0.
|
| 21 |
],
|
| 22 |
[
|
| 23 |
"model.language_model.layers.0.self_attn.o_proj.weight",
|
| 24 |
-
0.
|
| 25 |
],
|
| 26 |
[
|
| 27 |
"model.language_model.layers.0.mlp.gate_proj.weight",
|
| 28 |
-
0.
|
| 29 |
],
|
| 30 |
[
|
| 31 |
"model.language_model.layers.0.mlp.up_proj.weight",
|
| 32 |
-
0.
|
| 33 |
],
|
| 34 |
[
|
| 35 |
"model.language_model.layers.0.mlp.down_proj.weight",
|
|
@@ -37,7 +37,7 @@
|
|
| 37 |
],
|
| 38 |
[
|
| 39 |
"model.language_model.layers.1.self_attn.q_proj.weight",
|
| 40 |
-
0.
|
| 41 |
]
|
| 42 |
],
|
| 43 |
"file_sha256": {
|
|
@@ -45,8 +45,8 @@
|
|
| 45 |
"chat_template.jinja": "0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b",
|
| 46 |
"config.json": "c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3",
|
| 47 |
"generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
|
| 48 |
-
"model-00001-of-00002.safetensors": "
|
| 49 |
-
"model-00002-of-00002.safetensors": "
|
| 50 |
"model.safetensors.index.json": "55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e",
|
| 51 |
"params.json": "d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59",
|
| 52 |
"processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
|
|
@@ -55,5 +55,5 @@
|
|
| 55 |
"tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
|
| 56 |
"tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
|
| 57 |
},
|
| 58 |
-
"seconds":
|
| 59 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"snapshot": "mistralai/Ministral-3-3B-Instruct-2512-BF16@b6d637bef2393152b3da2b2fde72eecdee30557e",
|
| 3 |
+
"adapter": "StandardOne-3B-LoRA v2 (training run r5-3b)",
|
| 4 |
+
"adapter_sha256": "a8e3eb341e27c1a49a282e327c2a2038906abb0eee771e80debbb4a4c45f7afc",
|
| 5 |
"changed_params": 182,
|
| 6 |
"unchanged_params": 276,
|
| 7 |
"changed_outside_lm_projections": [],
|
| 8 |
+
"max_abs_delta": 0.002337932586669922,
|
| 9 |
"changed_sample": [
|
| 10 |
[
|
| 11 |
"model.language_model.layers.0.self_attn.q_proj.weight",
|
| 12 |
+
0.0010986328125
|
| 13 |
],
|
| 14 |
[
|
| 15 |
"model.language_model.layers.0.self_attn.k_proj.weight",
|
|
|
|
| 17 |
],
|
| 18 |
[
|
| 19 |
"model.language_model.layers.0.self_attn.v_proj.weight",
|
| 20 |
+
0.001434326171875
|
| 21 |
],
|
| 22 |
[
|
| 23 |
"model.language_model.layers.0.self_attn.o_proj.weight",
|
| 24 |
+
0.00103759765625
|
| 25 |
],
|
| 26 |
[
|
| 27 |
"model.language_model.layers.0.mlp.gate_proj.weight",
|
| 28 |
+
0.00200653076171875
|
| 29 |
],
|
| 30 |
[
|
| 31 |
"model.language_model.layers.0.mlp.up_proj.weight",
|
| 32 |
+
0.00177001953125
|
| 33 |
],
|
| 34 |
[
|
| 35 |
"model.language_model.layers.0.mlp.down_proj.weight",
|
|
|
|
| 37 |
],
|
| 38 |
[
|
| 39 |
"model.language_model.layers.1.self_attn.q_proj.weight",
|
| 40 |
+
0.00146484375
|
| 41 |
]
|
| 42 |
],
|
| 43 |
"file_sha256": {
|
|
|
|
| 45 |
"chat_template.jinja": "0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b",
|
| 46 |
"config.json": "c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3",
|
| 47 |
"generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
|
| 48 |
+
"model-00001-of-00002.safetensors": "7029a24151316b12d84b0e666c6af7d46bfe07e9549d59e513bb34afabb43073",
|
| 49 |
+
"model-00002-of-00002.safetensors": "ad184b56624eaca81a22c7fb7bef034edd25bd435cfb173a76d0db1b2c67ff2b",
|
| 50 |
"model.safetensors.index.json": "55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e",
|
| 51 |
"params.json": "d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59",
|
| 52 |
"processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
|
|
|
|
| 55 |
"tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
|
| 56 |
"tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
|
| 57 |
},
|
| 58 |
+
"seconds": 25.9
|
| 59 |
}
|
README.md
CHANGED
|
@@ -27,6 +27,12 @@ tags:
|
|
| 27 |
|
| 28 |
# Standard One 3B
|
| 29 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
|
| 31 |
`POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
|
| 32 |
BF16 3B checkpoint; the server code is in [StandardOne-8B](https://huggingface.co/StandardThinking/StandardOne-8B).
|
|
@@ -192,7 +198,7 @@ A JevBench v1.4.1 run has been requested; the sealed-set result is not yet avail
|
|
| 192 |
## Model details
|
| 193 |
|
| 194 |
- **Base model:** `mistralai/Ministral-3-3B-Instruct-2512-BF16`, revision `b6d637bef2393152b3da2b2fde72eecdee30557e` (Apache-2.0).
|
| 195 |
-
- **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **24,707,072** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 135,113,048 bytes, sha256 `
|
| 196 |
- **Merged BF16 checkpoint:** merging the adapter into the base changed **182 tensors**, none of them outside the language-model projections, maximum absolute weight change **0.0023**.
|
| 197 |
- **Serving details:** native chat-template wording, served without a system prompt, at a fixed temperature (T = 1.55, fitted on held-out calibration data); served model name `standard-one-3b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single, caller-supplied option order, no rotation ensemble; 8,192-token context.
|
| 198 |
|
|
@@ -207,9 +213,13 @@ The server code and full quick-start guide live in `StandardThinking/StandardOne
|
|
| 207 |
|
| 208 |
## Training data
|
| 209 |
|
| 210 |
-
Trains on the
|
|
|
|
|
|
|
|
|
|
|
|
|
| 211 |
|
| 212 |
-
An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.
|
| 213 |
|
| 214 |
## Limitations
|
| 215 |
|
|
|
|
| 27 |
|
| 28 |
# Standard One 3B
|
| 29 |
|
| 30 |
+
> **Updated weights (v2, 2026-09-26).** If you downloaded this model before, download it again or pin
|
| 31 |
+
> `revision="v2"`. Earlier versions stay available under the tags `v1` and `v1.1`.
|
| 32 |
+
> The benchmark figures and the serving temperature on this card are still those of v1.1 and are being updated for v2.
|
| 33 |
+
|
| 34 |
+
**Version:** v2
|
| 35 |
+
|
| 36 |
Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
|
| 37 |
`POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
|
| 38 |
BF16 3B checkpoint; the server code is in [StandardOne-8B](https://huggingface.co/StandardThinking/StandardOne-8B).
|
|
|
|
| 198 |
## Model details
|
| 199 |
|
| 200 |
- **Base model:** `mistralai/Ministral-3-3B-Instruct-2512-BF16`, revision `b6d637bef2393152b3da2b2fde72eecdee30557e` (Apache-2.0).
|
| 201 |
+
- **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **24,707,072** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 135,113,048 bytes, sha256 `a8e3eb341e27c1a49a282e327c2a2038906abb0eee771e80debbb4a4c45f7afc`.
|
| 202 |
- **Merged BF16 checkpoint:** merging the adapter into the base changed **182 tensors**, none of them outside the language-model projections, maximum absolute weight change **0.0023**.
|
| 203 |
- **Serving details:** native chat-template wording, served without a system prompt, at a fixed temperature (T = 1.55, fitted on held-out calibration data); served model name `standard-one-3b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single, caller-supplied option order, no rotation ensemble; 8,192-token context.
|
| 204 |
|
|
|
|
| 213 |
|
| 214 |
## Training data
|
| 215 |
|
| 216 |
+
Trains on the same data sources as `StandardThinking/StandardOne-8B`, not a reduced subset. Training data is synthetic and format-augmented decision data plus decision items converted from public
|
| 217 |
+
datasets (listed below); the JevBench public tiers used only for evaluation carry MIT. Full per-cohort breakdown: [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
|
| 218 |
+
|
| 219 |
+
Public datasets used (train splits where the dataset has one; licence as stated by each dataset; labels come from the
|
| 220 |
+
datasets, distractor options are generated by code): SQuAD 2.0 (CC BY-SA 4.0), ARC (CC BY-SA 4.0), BoolQ (CC BY-SA 3.0), CommonsenseQA (MIT), HellaSwag (MIT), Banking77 (CC BY 4.0), Bias in Bios (MIT), Bitext customer support (CDLA-Sharing-1.0), CLINC150 (CC BY 3.0), Amazon Counterfactual (CC BY 4.0), DBpedia-14 (CC BY-SA 3.0), Dolly 15k (CC BY-SA 3.0), GoEmotions (Apache-2.0), MASSIVE (CC BY 4.0), Twitter Financial News Sentiment (MIT), HelpSteer3 (CC BY 4.0), HelpSteer2 (CC BY 4.0), 2WikiMultihopQA (Apache-2.0), HotpotQA (CC BY-SA 4.0), MuSiQue (CC BY 4.0), QASC (CC BY 4.0), DROP (CC BY-SA 4.0), GSM8K (MIT), TempReason (CC BY-SA 3.0), MultiNLI (OANC / CC BY-SA 3.0 / CC BY 3.0), PAWS (Google terms, free for any purpose), PAWS-X (Google terms, free for any purpose), SNLI (CC BY-SA 4.0), WANLI (CC BY 4.0), ContractNLI (CC BY 4.0), CUAD (CC BY 4.0), ShARC (CC BY-SA 3.0), Jailbreak classification (Apache-2.0), Prompt injections (Apache-2.0), Aegis AI Content Safety 2.0 (CC BY 4.0), Jigsaw Toxic Comment Classification (mirror of the Kaggle data) (CC0 (data); comment text CC BY-SA 3.0 (Wikipedia)), Measuring Hate Speech (CC BY 4.0), Image safety classes (MIT). Upstream ids and the cohort each one feeds: [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
|
| 221 |
|
| 222 |
+
An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.03 % of the 520,754-row training mixture). These rows are kept and disclosed here rather than regenerated, since the overlap is limited to one rubric question's wording and never touches a scenario or an answer.
|
| 223 |
|
| 224 |
## Limitations
|
| 225 |
|
SHA256SUMS
CHANGED
|
@@ -1,12 +1,12 @@
|
|
| 1 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
|
| 2 |
-
|
| 3 |
f10343f951ecb3a72670e12720de6e04d59e9cb872e84b2ceddd7d54853bb7e7 NOTICE
|
| 4 |
372d9160e95493d1d8e5c94d82b12fb0693fd36e4a50f5206d287ee74f189046 QUICKSTART.md
|
| 5 |
-
|
| 6 |
331b249682cd52226c50e533f59825184997f3b31f9a62b2c3fea940db6999c5 SYSTEM_PROMPT.txt
|
| 7 |
0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b chat_template.jinja
|
| 8 |
c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3 config.json
|
| 9 |
-
|
| 10 |
96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
|
| 11 |
28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
|
| 12 |
7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
|
|
@@ -25,12 +25,12 @@ dd888be78c0e32894cf04a8e3fbf8757e7218873668fa4443ba47c7c1c48b6dc docs/assets/04
|
|
| 25 |
7b9b1ee2f93166a817ead61bbc0f4c62984afe7d6c8affefd051007582d0a485 evidence/served-nosys-latency-idle-gpu.json
|
| 26 |
5b217fe780b4c8c7cb292d57630b0f0a0bb5dcbe275f731272d6f0804cc14a68 evidence/served-nosys-original-report.json
|
| 27 |
e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
|
| 28 |
-
|
| 29 |
-
|
| 30 |
55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e model.safetensors.index.json
|
| 31 |
d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59 params.json
|
| 32 |
ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
|
| 33 |
-
|
| 34 |
eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
|
| 35 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
|
| 36 |
0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
|
|
|
|
| 1 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
|
| 2 |
+
916927708f20ec40252055f98ad7f09f5dcfacadc31e99d85bcac19f285b7bad MERGE_REPORT.json
|
| 3 |
f10343f951ecb3a72670e12720de6e04d59e9cb872e84b2ceddd7d54853bb7e7 NOTICE
|
| 4 |
372d9160e95493d1d8e5c94d82b12fb0693fd36e4a50f5206d287ee74f189046 QUICKSTART.md
|
| 5 |
+
faf66b200ab0a00f0e7f7c7174af1996c45376d3e529ed08ea7582b232c1e602 README.md
|
| 6 |
331b249682cd52226c50e533f59825184997f3b31f9a62b2c3fea940db6999c5 SYSTEM_PROMPT.txt
|
| 7 |
0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b chat_template.jinja
|
| 8 |
c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3 config.json
|
| 9 |
+
141067bb117600df9e09a46098e6a1f26e685ea5eab77f73b20352bce6d503d6 docs/BENCHMARKS.md
|
| 10 |
96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
|
| 11 |
28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
|
| 12 |
7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
|
|
|
|
| 25 |
7b9b1ee2f93166a817ead61bbc0f4c62984afe7d6c8affefd051007582d0a485 evidence/served-nosys-latency-idle-gpu.json
|
| 26 |
5b217fe780b4c8c7cb292d57630b0f0a0bb5dcbe275f731272d6f0804cc14a68 evidence/served-nosys-original-report.json
|
| 27 |
e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
|
| 28 |
+
7029a24151316b12d84b0e666c6af7d46bfe07e9549d59e513bb34afabb43073 model-00001-of-00002.safetensors
|
| 29 |
+
ad184b56624eaca81a22c7fb7bef034edd25bd435cfb173a76d0db1b2c67ff2b model-00002-of-00002.safetensors
|
| 30 |
55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e model.safetensors.index.json
|
| 31 |
d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59 params.json
|
| 32 |
ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
|
| 33 |
+
f9913e0d49fc053df656058d13af2523b6263f9396efad7a6fa06e7a121e1acf release-manifest.json
|
| 34 |
eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
|
| 35 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
|
| 36 |
0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
|
docs/BENCHMARKS.md
CHANGED
|
@@ -1,5 +1,8 @@
|
|
| 1 |
# Standard One — full benchmark report
|
| 2 |
|
|
|
|
|
|
|
|
|
|
| 3 |
This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
|
| 4 |
**Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
|
| 5 |
leaderboard score. The only comparisons in this report are: the untuned
|
|
@@ -280,9 +283,8 @@ and in `release-manifest.json`, identical for both candidates.
|
|
| 280 |
|
| 281 |
## Training data provenance
|
| 282 |
|
| 283 |
-
The following cohort families make up the
|
| 284 |
-
both Standard One 8B and Standard One 3B.
|
| 285 |
-
visible checkers; a subset was teacher-reviewed). No JevBench evaluation scenario appears in any
|
| 286 |
training cohort.
|
| 287 |
|
| 288 |
| Cohort family | What it covers | Licence |
|
|
@@ -295,11 +297,57 @@ training cohort.
|
|
| 295 |
| Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
|
| 296 |
| Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
|
| 297 |
| Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
|
|
|
|
|
|
|
| 298 |
| JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
|
| 299 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 300 |
## Public-overlap audit and disclosure
|
| 301 |
|
| 302 |
-
Run on the
|
| 303 |
|
| 304 |
- **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
|
| 305 |
- **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
|
|
@@ -307,13 +355,10 @@ Run on the exact final 34-cohort, 382,576-row mixture that **both candidates tra
|
|
| 307 |
one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
|
| 308 |
response fully and correctly satisfy the request?"), reused as the complete instruction rather than
|
| 309 |
as a substring.
|
| 310 |
-
-
|
| 311 |
-
|
| 312 |
-
`judge-proxy-v1`), an insurance "cancellation takes effect" boilerplate clause, and a "coverage
|
| 313 |
-
reviewer" claims-adjudication phrase.
|
| 314 |
-
- All other 32 cohorts: 0 exact state and 0 exact instruction matches.
|
| 315 |
|
| 316 |
-
**These 181 rows are kept and disclosed here, not regenerated.** They are 0.
|
| 317 |
the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
|
| 318 |
|
| 319 |
## Evidence file hashes
|
|
|
|
| 1 |
# Standard One — full benchmark report
|
| 2 |
|
| 3 |
+
> **Note.** The benchmark figures in this report are for v1.1 and are being updated for v2. The training-data provenance and
|
| 4 |
+
> overlap-audit sections below already describe the v2 weights.
|
| 5 |
+
|
| 6 |
This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
|
| 7 |
**Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
|
| 8 |
leaderboard score. The only comparisons in this report are: the untuned
|
|
|
|
| 283 |
|
| 284 |
## Training data provenance
|
| 285 |
|
| 286 |
+
The following cohort families make up the 46-cohort, 520,754-row training mixture used by
|
| 287 |
+
both Standard One 8B and Standard One 3B. Apart from the public-dataset items listed below, every row is synthetic (deterministic generators with visible checkers; a subset was teacher-reviewed) or format-augmented. No JevBench evaluation scenario appears in any
|
|
|
|
| 288 |
training cohort.
|
| 289 |
|
| 290 |
| Cohort family | What it covers | Licence |
|
|
|
|
| 297 |
| Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
|
| 298 |
| Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
|
| 299 |
| Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
|
| 300 |
+
| Synthetic and format-augmented decision data (v2) | Decision items in varied layouts | Apache-2.0 (ours) |
|
| 301 |
+
| Public-dataset decision items | Decision items converted from public datasets (table below) | Each dataset's own licence (table below) |
|
| 302 |
| JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
|
| 303 |
|
| 304 |
+
Public datasets used (train splits where a dataset has one; CUAD: its single release file). Labels come from the datasets;
|
| 305 |
+
distractor options and derived labels are generated by code. Licences as stated by each dataset; attribution to the dataset authors.
|
| 306 |
+
|
| 307 |
+
| Dataset | Upstream id | Licence | Cohort |
|
| 308 |
+
|---|---|---|---|
|
| 309 |
+
| SQuAD 2.0 | `rajpurkar/squad_v2` | CC BY-SA 4.0 | `pub-abstain-v1` |
|
| 310 |
+
| ARC | `allenai/ai2_arc` | CC BY-SA 4.0 | `pub-breadth-v1` |
|
| 311 |
+
| BoolQ | `google/boolq` | CC BY-SA 3.0 | `pub-breadth-v1` |
|
| 312 |
+
| CommonsenseQA | `tau/commonsense_qa` | MIT | `pub-breadth-v1` |
|
| 313 |
+
| HellaSwag | `Rowan/hellaswag` | MIT | `pub-breadth-v1` |
|
| 314 |
+
| Banking77 | `PolyAI/banking77` | CC BY 4.0 | `pub-classify-v1` |
|
| 315 |
+
| Bias in Bios | `LabHC/bias_in_bios` | MIT | `pub-classify-v1` |
|
| 316 |
+
| Bitext customer support | `bitext/Bitext-customer-support-llm-chatbot-training-dataset` | CDLA-Sharing-1.0 | `pub-classify-v1` |
|
| 317 |
+
| CLINC150 | `clinc/clinc_oos` | CC BY 3.0 | `pub-classify-v1` |
|
| 318 |
+
| Amazon Counterfactual | `mteb/amazon_counterfactual` | CC BY 4.0 | `pub-classify-v1` |
|
| 319 |
+
| DBpedia-14 | `fancyzhx/dbpedia_14` | CC BY-SA 3.0 | `pub-classify-v1` |
|
| 320 |
+
| Dolly 15k | `databricks/databricks-dolly-15k` | CC BY-SA 3.0 | `pub-classify-v1`, `safety-judge-v1` |
|
| 321 |
+
| GoEmotions | `google-research-datasets/go_emotions` | Apache-2.0 | `pub-classify-v1` |
|
| 322 |
+
| MASSIVE | `AmazonScience/massive` | CC BY 4.0 | `pub-classify-v1` |
|
| 323 |
+
| Twitter Financial News Sentiment | `zeroshot/twitter-financial-news-sentiment` | MIT | `pub-classify-v1` |
|
| 324 |
+
| HelpSteer3 | `nvidia/HelpSteer3` | CC BY 4.0 | `pub-judge-v1` |
|
| 325 |
+
| HelpSteer2 | `nvidia/HelpSteer2` | CC BY 4.0 | `pub-judge-v1` |
|
| 326 |
+
| 2WikiMultihopQA | `framolfese/2WikiMultihopQA` | Apache-2.0 | `pub-multihop-v1` |
|
| 327 |
+
| HotpotQA | `hotpotqa/hotpot_qa` | CC BY-SA 4.0 | `pub-multihop-v1` |
|
| 328 |
+
| MuSiQue | `dgslibisey/MuSiQue` | CC BY 4.0 | `pub-multihop-v1` |
|
| 329 |
+
| QASC | `allenai/qasc` | CC BY 4.0 | `pub-multihop-v1` |
|
| 330 |
+
| DROP | `ucinlp/drop` | CC BY-SA 4.0 | `pub-numeric-v1` |
|
| 331 |
+
| GSM8K | `openai/gsm8k` | MIT | `pub-numeric-v1` |
|
| 332 |
+
| TempReason | `tonytan48/TempReason` | CC BY-SA 3.0 | `pub-numeric-v1` |
|
| 333 |
+
| MultiNLI | `nyu-mll/multi_nli` | OANC / CC BY-SA 3.0 / CC BY 3.0 | `pub-paraphrase-v1` |
|
| 334 |
+
| PAWS | `google-research-datasets/paws` | Google terms, free for any purpose | `pub-paraphrase-v1` |
|
| 335 |
+
| PAWS-X | `google-research-datasets/paws-x` | Google terms, free for any purpose | `pub-paraphrase-v1` |
|
| 336 |
+
| SNLI | `stanfordnlp/snli` | CC BY-SA 4.0 | `pub-paraphrase-v1` |
|
| 337 |
+
| WANLI | `alisawuffles/WANLI` | CC BY 4.0 | `pub-paraphrase-v1` |
|
| 338 |
+
| ContractNLI | `stanfordnlp.github.io/contract-nli` | CC BY 4.0 | `pub-rules-v1` |
|
| 339 |
+
| CUAD | `theatticusproject/cuad` | CC BY 4.0 | `pub-rules-v1` |
|
| 340 |
+
| ShARC | `sharc-data.github.io` | CC BY-SA 3.0 | `pub-rules-v1` |
|
| 341 |
+
| Jailbreak classification | `jackhhao/jailbreak-classification` | Apache-2.0 | `safety-judge-v1` |
|
| 342 |
+
| Prompt injections | `deepset/prompt-injections` | Apache-2.0 | `safety-judge-v1` |
|
| 343 |
+
| Aegis AI Content Safety 2.0 | `nvidia/Aegis-AI-Content-Safety-Dataset-2.0` | CC BY 4.0 | `content-safety-text-v1` |
|
| 344 |
+
| Jigsaw Toxic Comment Classification (mirror of the Kaggle data) | `tasksource/jigsaw_toxicity` | CC0 (data); comment text CC BY-SA 3.0 (Wikipedia) | `content-safety-text-v1` |
|
| 345 |
+
| Measuring Hate Speech | `ucberkeley-dlab/measuring-hate-speech` | CC BY 4.0 | `content-safety-text-v1` |
|
| 346 |
+
| Image safety classes | `deepghs/nsfw_detect` | MIT | `content-safety-image-v1`, `content-safety-image-2d-v1` |
|
| 347 |
+
|
| 348 |
## Public-overlap audit and disclosure
|
| 349 |
|
| 350 |
+
Run on the 46-cohort, 520,754-row training data that **both candidates train on**:
|
| 351 |
|
| 352 |
- **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
|
| 353 |
- **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
|
|
|
|
| 355 |
one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
|
| 356 |
response fully and correctly satisfy the request?"), reused as the complete instruction rather than
|
| 357 |
as a substring.
|
| 358 |
+
- 24 distinct shared word 8-grams (20,400 rows), all in known families: the rubric sentence above (also in `judge-realistic-train-v3`, `judge-multilingual-train-v1`, `judge-proxy-v1` and the format-augmented cohort), an insurance "cancellation takes effect" boilerplate clause, a "coverage reviewer" claims-adjudication phrase, and generic contract boilerplate (for example "controlled by or under common control with") in one public-dataset cohort.
|
| 359 |
+
- All other 44 cohorts: 0 exact state and 0 exact instruction matches.
|
|
|
|
|
|
|
|
|
|
| 360 |
|
| 361 |
+
**These 181 rows are kept and disclosed here, not regenerated.** They are 0.03 % of the mixture, and
|
| 362 |
the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
|
| 363 |
|
| 364 |
## Evidence file hashes
|
model-00001-of-00002.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4995664864
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7029a24151316b12d84b0e666c6af7d46bfe07e9549d59e513bb34afabb43073
|
| 3 |
size 4995664864
|
model-00002-of-00002.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2702576200
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ad184b56624eaca81a22c7fb7bef034edd25bd435cfb173a76d0db1b2c67ff2b
|
| 3 |
size 2702576200
|
release-manifest.json
CHANGED
|
@@ -1,24 +1,26 @@
|
|
| 1 |
{
|
| 2 |
"release": "StandardOne-3B",
|
|
|
|
| 3 |
"base_model": {
|
| 4 |
"repo": "mistralai/Ministral-3-3B-Instruct-2512-BF16",
|
| 5 |
"revision": "b6d637bef2393152b3da2b2fde72eecdee30557e"
|
| 6 |
},
|
| 7 |
"adapter": {
|
| 8 |
-
"source_run": "
|
| 9 |
-
"adapter_model_sha256": "
|
| 10 |
},
|
| 11 |
"merge_report_summary": {
|
| 12 |
"changed_params": 182,
|
| 13 |
"unchanged_params": 276,
|
| 14 |
"changed_outside_lm_projections": [],
|
| 15 |
-
"max_abs_delta": 0.
|
| 16 |
},
|
| 17 |
"serving": {
|
| 18 |
"wording": "native",
|
| 19 |
"system_prompt": null,
|
| 20 |
"temperature": 1.55,
|
| 21 |
"served_model_name": "standard-one-3b",
|
| 22 |
-
"engine": "SGLang 0.5.20"
|
|
|
|
| 23 |
}
|
| 24 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"release": "StandardOne-3B",
|
| 3 |
+
"version": "v2",
|
| 4 |
"base_model": {
|
| 5 |
"repo": "mistralai/Ministral-3-3B-Instruct-2512-BF16",
|
| 6 |
"revision": "b6d637bef2393152b3da2b2fde72eecdee30557e"
|
| 7 |
},
|
| 8 |
"adapter": {
|
| 9 |
+
"source_run": "r5-3b",
|
| 10 |
+
"adapter_model_sha256": "a8e3eb341e27c1a49a282e327c2a2038906abb0eee771e80debbb4a4c45f7afc"
|
| 11 |
},
|
| 12 |
"merge_report_summary": {
|
| 13 |
"changed_params": 182,
|
| 14 |
"unchanged_params": 276,
|
| 15 |
"changed_outside_lm_projections": [],
|
| 16 |
+
"max_abs_delta": 0.002337932586669922
|
| 17 |
},
|
| 18 |
"serving": {
|
| 19 |
"wording": "native",
|
| 20 |
"system_prompt": null,
|
| 21 |
"temperature": 1.55,
|
| 22 |
"served_model_name": "standard-one-3b",
|
| 23 |
+
"engine": "SGLang 0.5.20",
|
| 24 |
+
"note": "serving settings shown are those of v1.1; they are being re-fitted for v2"
|
| 25 |
}
|
| 26 |
}
|