Text Generation
Transformers
Safetensors
mistral3
image-text-to-text
decision-model
typed-decisions
jev
jevbench
calibration
decode-free
multilingual
vision-language
conversational
Instructions to use StandardThinking/StandardOne-8B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use StandardThinking/StandardOne-8B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="StandardThinking/StandardOne-8B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("StandardThinking/StandardOne-8B") model = AutoModelForMultimodalLM.from_pretrained("StandardThinking/StandardOne-8B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use StandardThinking/StandardOne-8B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "StandardThinking/StandardOne-8B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-8B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/StandardThinking/StandardOne-8B
- SGLang
How to use StandardThinking/StandardOne-8B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-8B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-8B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-8B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-8B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use StandardThinking/StandardOne-8B with Docker Model Runner:
docker model run hf.co/StandardThinking/StandardOne-8B
Update weights
Browse files- MERGE_REPORT.json +13 -13
- README.md +14 -4
- SHA256SUMS +8 -8
- docs/BENCHMARKS.md +55 -10
- model-00001-of-00004.safetensors +1 -1
- model-00002-of-00004.safetensors +1 -1
- model-00003-of-00004.safetensors +1 -1
- model-00004-of-00004.safetensors +1 -1
- release-manifest.json +6 -4
MERGE_REPORT.json
CHANGED
|
@@ -1,11 +1,11 @@
|
|
| 1 |
{
|
| 2 |
"snapshot": "mistralai/Ministral-3-8B-Instruct-2512-BF16@f6fae9795746f63c9be8344932f01275f3c63734",
|
| 3 |
-
"adapter": "StandardOne-8B-LoRA (training run
|
| 4 |
-
"adapter_sha256": "
|
| 5 |
"changed_params": 238,
|
| 6 |
"unchanged_params": 293,
|
| 7 |
"changed_outside_lm_projections": [],
|
| 8 |
-
"max_abs_delta": 0.
|
| 9 |
"changed_sample": [
|
| 10 |
[
|
| 11 |
"model.language_model.layers.0.self_attn.q_proj.weight",
|
|
@@ -17,23 +17,23 @@
|
|
| 17 |
],
|
| 18 |
[
|
| 19 |
"model.language_model.layers.0.self_attn.v_proj.weight",
|
| 20 |
-
0.
|
| 21 |
],
|
| 22 |
[
|
| 23 |
"model.language_model.layers.0.self_attn.o_proj.weight",
|
| 24 |
-
0.
|
| 25 |
],
|
| 26 |
[
|
| 27 |
"model.language_model.layers.0.mlp.gate_proj.weight",
|
| 28 |
-
0.
|
| 29 |
],
|
| 30 |
[
|
| 31 |
"model.language_model.layers.0.mlp.up_proj.weight",
|
| 32 |
-
0.
|
| 33 |
],
|
| 34 |
[
|
| 35 |
"model.language_model.layers.0.mlp.down_proj.weight",
|
| 36 |
-
0.
|
| 37 |
],
|
| 38 |
[
|
| 39 |
"model.language_model.layers.1.self_attn.q_proj.weight",
|
|
@@ -45,10 +45,10 @@
|
|
| 45 |
"chat_template.jinja": "74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56",
|
| 46 |
"config.json": "111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882",
|
| 47 |
"generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
|
| 48 |
-
"model-00001-of-00004.safetensors": "
|
| 49 |
-
"model-00002-of-00004.safetensors": "
|
| 50 |
-
"model-00003-of-00004.safetensors": "
|
| 51 |
-
"model-00004-of-00004.safetensors": "
|
| 52 |
"model.safetensors.index.json": "71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb",
|
| 53 |
"params.json": "81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538",
|
| 54 |
"processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
|
|
@@ -57,5 +57,5 @@
|
|
| 57 |
"tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
|
| 58 |
"tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
|
| 59 |
},
|
| 60 |
-
"seconds":
|
| 61 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"snapshot": "mistralai/Ministral-3-8B-Instruct-2512-BF16@f6fae9795746f63c9be8344932f01275f3c63734",
|
| 3 |
+
"adapter": "StandardOne-8B-LoRA v2 (training run r5-8b)",
|
| 4 |
+
"adapter_sha256": "53e41238cd55567cfbc75efdf61d354da39771573f56f74bb1440c9ffdd1bd4a",
|
| 5 |
"changed_params": 238,
|
| 6 |
"unchanged_params": 293,
|
| 7 |
"changed_outside_lm_projections": [],
|
| 8 |
+
"max_abs_delta": 0.00201416015625,
|
| 9 |
"changed_sample": [
|
| 10 |
[
|
| 11 |
"model.language_model.layers.0.self_attn.q_proj.weight",
|
|
|
|
| 17 |
],
|
| 18 |
[
|
| 19 |
"model.language_model.layers.0.self_attn.v_proj.weight",
|
| 20 |
+
0.0011348724365234375
|
| 21 |
],
|
| 22 |
[
|
| 23 |
"model.language_model.layers.0.self_attn.o_proj.weight",
|
| 24 |
+
0.001430511474609375
|
| 25 |
],
|
| 26 |
[
|
| 27 |
"model.language_model.layers.0.mlp.gate_proj.weight",
|
| 28 |
+
0.00112152099609375
|
| 29 |
],
|
| 30 |
[
|
| 31 |
"model.language_model.layers.0.mlp.up_proj.weight",
|
| 32 |
+
0.001220703125
|
| 33 |
],
|
| 34 |
[
|
| 35 |
"model.language_model.layers.0.mlp.down_proj.weight",
|
| 36 |
+
0.00058746337890625
|
| 37 |
],
|
| 38 |
[
|
| 39 |
"model.language_model.layers.1.self_attn.q_proj.weight",
|
|
|
|
| 45 |
"chat_template.jinja": "74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56",
|
| 46 |
"config.json": "111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882",
|
| 47 |
"generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
|
| 48 |
+
"model-00001-of-00004.safetensors": "5b29a2e3e86366c0b2e21c5038a16fa1f2bed8a2ca69e5e775608126570c89d4",
|
| 49 |
+
"model-00002-of-00004.safetensors": "65c9000a4aafa0cb0978540cfdbb6017e39f14a208a8d0523d375656f5cf0343",
|
| 50 |
+
"model-00003-of-00004.safetensors": "fe537b22d9b005873c25b0f8ca0548e24f6d3a4fb611db11becc6ba86c849a03",
|
| 51 |
+
"model-00004-of-00004.safetensors": "6c3f84685c69360d991bf02b81ac670fcb36cc1b9af2dd51e2137e0f3d36a24a",
|
| 52 |
"model.safetensors.index.json": "71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb",
|
| 53 |
"params.json": "81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538",
|
| 54 |
"processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
|
|
|
|
| 57 |
"tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
|
| 58 |
"tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
|
| 59 |
},
|
| 60 |
+
"seconds": 65.5
|
| 61 |
}
|
README.md
CHANGED
|
@@ -27,6 +27,12 @@ tags:
|
|
| 27 |
|
| 28 |
# Standard One 8B
|
| 29 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
|
| 31 |
`POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
|
| 32 |
BF16 8B checkpoint and the server code.
|
|
@@ -192,8 +198,8 @@ A JevBench v1.4.1 run has been requested; the sealed-set result is not yet avail
|
|
| 192 |
## Model details
|
| 193 |
|
| 194 |
- **Base model:** `mistralai/Ministral-3-8B-Instruct-2512-BF16`, revision `f6fae9795746f63c9be8344932f01275f3c63734` (Apache-2.0).
|
| 195 |
-
- **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **44,564,480** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 214,559,872 bytes, sha256 `
|
| 196 |
-
- **Merged BF16 checkpoint:** merging the adapter into the base changed **238 tensors** (293 unchanged), none outside the language-model projections, max absolute weight change **0.
|
| 197 |
- **Serving details:** native chat-template wording, no system prompt, fixed temperature (T = 1.65, fitted on held-out calibration data); served model name `standard-one-8b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single caller-supplied option order, no rotation ensemble; 8,192-token context.
|
| 198 |
|
| 199 |
| Path | Contents |
|
|
@@ -206,9 +212,13 @@ A JevBench v1.4.1 run has been requested; the sealed-set result is not yet avail
|
|
| 206 |
|
| 207 |
## Training data
|
| 208 |
|
| 209 |
-
Training data
|
|
|
|
|
|
|
|
|
|
|
|
|
| 210 |
|
| 211 |
-
An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.
|
| 212 |
|
| 213 |
## Limitations
|
| 214 |
|
|
|
|
| 27 |
|
| 28 |
# Standard One 8B
|
| 29 |
|
| 30 |
+
> **Updated weights (v2, 2026-09-26).** If you downloaded this model before, download it again or pin
|
| 31 |
+
> `revision="v2"`. Earlier versions stay available under the tags `v1` and `v1.1`.
|
| 32 |
+
> The benchmark figures and the serving temperature on this card are still those of v1.1 and are being updated for v2.
|
| 33 |
+
|
| 34 |
+
**Version:** v2
|
| 35 |
+
|
| 36 |
Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
|
| 37 |
`POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
|
| 38 |
BF16 8B checkpoint and the server code.
|
|
|
|
| 198 |
## Model details
|
| 199 |
|
| 200 |
- **Base model:** `mistralai/Ministral-3-8B-Instruct-2512-BF16`, revision `f6fae9795746f63c9be8344932f01275f3c63734` (Apache-2.0).
|
| 201 |
+
- **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **44,564,480** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 214,559,872 bytes, sha256 `53e41238cd55567cfbc75efdf61d354da39771573f56f74bb1440c9ffdd1bd4a`.
|
| 202 |
+
- **Merged BF16 checkpoint:** merging the adapter into the base changed **238 tensors** (293 unchanged), none outside the language-model projections, max absolute weight change **0.00201**.
|
| 203 |
- **Serving details:** native chat-template wording, no system prompt, fixed temperature (T = 1.65, fitted on held-out calibration data); served model name `standard-one-8b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single caller-supplied option order, no rotation ensemble; 8,192-token context.
|
| 204 |
|
| 205 |
| Path | Contents |
|
|
|
|
| 212 |
|
| 213 |
## Training data
|
| 214 |
|
| 215 |
+
Training data is synthetic and format-augmented decision data plus decision items converted from public
|
| 216 |
+
datasets (listed below); the JevBench public tiers used only for evaluation carry MIT. Full per-cohort breakdown (row counts, what each covers, licence): [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
|
| 217 |
+
|
| 218 |
+
Public datasets used (train splits where the dataset has one; licence as stated by each dataset; labels come from the
|
| 219 |
+
datasets, distractor options are generated by code): SQuAD 2.0 (CC BY-SA 4.0), ARC (CC BY-SA 4.0), BoolQ (CC BY-SA 3.0), CommonsenseQA (MIT), HellaSwag (MIT), Banking77 (CC BY 4.0), Bias in Bios (MIT), Bitext customer support (CDLA-Sharing-1.0), CLINC150 (CC BY 3.0), Amazon Counterfactual (CC BY 4.0), DBpedia-14 (CC BY-SA 3.0), Dolly 15k (CC BY-SA 3.0), GoEmotions (Apache-2.0), MASSIVE (CC BY 4.0), Twitter Financial News Sentiment (MIT), HelpSteer3 (CC BY 4.0), HelpSteer2 (CC BY 4.0), 2WikiMultihopQA (Apache-2.0), HotpotQA (CC BY-SA 4.0), MuSiQue (CC BY 4.0), QASC (CC BY 4.0), DROP (CC BY-SA 4.0), GSM8K (MIT), TempReason (CC BY-SA 3.0), MultiNLI (OANC / CC BY-SA 3.0 / CC BY 3.0), PAWS (Google terms, free for any purpose), PAWS-X (Google terms, free for any purpose), SNLI (CC BY-SA 4.0), WANLI (CC BY 4.0), ContractNLI (CC BY 4.0), CUAD (CC BY 4.0), ShARC (CC BY-SA 3.0), Jailbreak classification (Apache-2.0), Prompt injections (Apache-2.0), Aegis AI Content Safety 2.0 (CC BY 4.0), Jigsaw Toxic Comment Classification (mirror of the Kaggle data) (CC0 (data); comment text CC BY-SA 3.0 (Wikipedia)), Measuring Hate Speech (CC BY 4.0), Image safety classes (MIT). Upstream ids and the cohort each one feeds: [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
|
| 220 |
|
| 221 |
+
An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.03 % of the 520,754-row training mixture). These rows are kept and disclosed here rather than regenerated, since the overlap is limited to one rubric question's wording and never touches a scenario or an answer.
|
| 222 |
|
| 223 |
## Limitations
|
| 224 |
|
SHA256SUMS
CHANGED
|
@@ -1,12 +1,12 @@
|
|
| 1 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
|
| 2 |
-
|
| 3 |
20486cf6b2e42440b0aea913ea475c818664f34e76e4501651d6cb21987a508b NOTICE
|
| 4 |
de8701f5a28091cdbc601d0625769a1356c6e31962df4fc2c2a559fe4a67369f QUICKSTART.md
|
| 5 |
-
|
| 6 |
66e0fda8c8ab269ccd010dca6196931d475a22c3f2eebe5b6b3a2ec33a354d0e SYSTEM_PROMPT.txt
|
| 7 |
74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56 chat_template.jinja
|
| 8 |
111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882 config.json
|
| 9 |
-
|
| 10 |
96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
|
| 11 |
28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
|
| 12 |
7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
|
|
@@ -25,14 +25,14 @@ c0eb0fde38c44b1f3f62f6a79cc45c3e89cad1f782d1c6cc40526f69e8595ff6 evidence/serve
|
|
| 25 |
d47f43be5bd7767b80ed27a96a9a85ece1495b37fb9365e575b1df6b32fbc509 evidence/served-nosys-latency-idle-gpu.json
|
| 26 |
b87e0a9a48ebb9fa933fe0db470b506cbe1850824abf65dfebf6ad6c5f58cdda evidence/served-nosys-original-report.json
|
| 27 |
e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
|
| 28 |
-
|
| 29 |
-
|
| 30 |
-
|
| 31 |
-
|
| 32 |
71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb model.safetensors.index.json
|
| 33 |
81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538 params.json
|
| 34 |
ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
|
| 35 |
-
|
| 36 |
eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
|
| 37 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
|
| 38 |
0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
|
|
|
|
| 1 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
|
| 2 |
+
0fae9e8380c795cd6d6fa9f1733c408df7b27bf009d947ebb15adc96ad40dfab MERGE_REPORT.json
|
| 3 |
20486cf6b2e42440b0aea913ea475c818664f34e76e4501651d6cb21987a508b NOTICE
|
| 4 |
de8701f5a28091cdbc601d0625769a1356c6e31962df4fc2c2a559fe4a67369f QUICKSTART.md
|
| 5 |
+
ebcf3db9b80fb72f46b2616ed24ac37f10610eb7173f98f6247054d76da48bac README.md
|
| 6 |
66e0fda8c8ab269ccd010dca6196931d475a22c3f2eebe5b6b3a2ec33a354d0e SYSTEM_PROMPT.txt
|
| 7 |
74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56 chat_template.jinja
|
| 8 |
111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882 config.json
|
| 9 |
+
141067bb117600df9e09a46098e6a1f26e685ea5eab77f73b20352bce6d503d6 docs/BENCHMARKS.md
|
| 10 |
96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
|
| 11 |
28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
|
| 12 |
7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
|
|
|
|
| 25 |
d47f43be5bd7767b80ed27a96a9a85ece1495b37fb9365e575b1df6b32fbc509 evidence/served-nosys-latency-idle-gpu.json
|
| 26 |
b87e0a9a48ebb9fa933fe0db470b506cbe1850824abf65dfebf6ad6c5f58cdda evidence/served-nosys-original-report.json
|
| 27 |
e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
|
| 28 |
+
5b29a2e3e86366c0b2e21c5038a16fa1f2bed8a2ca69e5e775608126570c89d4 model-00001-of-00004.safetensors
|
| 29 |
+
65c9000a4aafa0cb0978540cfdbb6017e39f14a208a8d0523d375656f5cf0343 model-00002-of-00004.safetensors
|
| 30 |
+
fe537b22d9b005873c25b0f8ca0548e24f6d3a4fb611db11becc6ba86c849a03 model-00003-of-00004.safetensors
|
| 31 |
+
6c3f84685c69360d991bf02b81ac670fcb36cc1b9af2dd51e2137e0f3d36a24a model-00004-of-00004.safetensors
|
| 32 |
71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb model.safetensors.index.json
|
| 33 |
81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538 params.json
|
| 34 |
ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
|
| 35 |
+
2eb900bdf73188c24ad9e8c0d1f47aadf0b6531eb33a0ce8929b52914cc183b3 release-manifest.json
|
| 36 |
eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
|
| 37 |
1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
|
| 38 |
0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
|
docs/BENCHMARKS.md
CHANGED
|
@@ -1,5 +1,8 @@
|
|
| 1 |
# Standard One — full benchmark report
|
| 2 |
|
|
|
|
|
|
|
|
|
|
| 3 |
This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
|
| 4 |
**Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
|
| 5 |
leaderboard score. The only comparisons in this report are: the untuned
|
|
@@ -280,9 +283,8 @@ and in `release-manifest.json`, identical for both candidates.
|
|
| 280 |
|
| 281 |
## Training data provenance
|
| 282 |
|
| 283 |
-
The following cohort families make up the
|
| 284 |
-
both Standard One 8B and Standard One 3B.
|
| 285 |
-
visible checkers; a subset was teacher-reviewed). No JevBench evaluation scenario appears in any
|
| 286 |
training cohort.
|
| 287 |
|
| 288 |
| Cohort family | What it covers | Licence |
|
|
@@ -295,11 +297,57 @@ training cohort.
|
|
| 295 |
| Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
|
| 296 |
| Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
|
| 297 |
| Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
|
|
|
|
|
|
|
| 298 |
| JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
|
| 299 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 300 |
## Public-overlap audit and disclosure
|
| 301 |
|
| 302 |
-
Run on the
|
| 303 |
|
| 304 |
- **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
|
| 305 |
- **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
|
|
@@ -307,13 +355,10 @@ Run on the exact final 34-cohort, 382,576-row mixture that **both candidates tra
|
|
| 307 |
one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
|
| 308 |
response fully and correctly satisfy the request?"), reused as the complete instruction rather than
|
| 309 |
as a substring.
|
| 310 |
-
-
|
| 311 |
-
|
| 312 |
-
`judge-proxy-v1`), an insurance "cancellation takes effect" boilerplate clause, and a "coverage
|
| 313 |
-
reviewer" claims-adjudication phrase.
|
| 314 |
-
- All other 32 cohorts: 0 exact state and 0 exact instruction matches.
|
| 315 |
|
| 316 |
-
**These 181 rows are kept and disclosed here, not regenerated.** They are 0.
|
| 317 |
the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
|
| 318 |
|
| 319 |
## Evidence file hashes
|
|
|
|
| 1 |
# Standard One — full benchmark report
|
| 2 |
|
| 3 |
+
> **Note.** The benchmark figures in this report are for v1.1 and are being updated for v2. The training-data provenance and
|
| 4 |
+
> overlap-audit sections below already describe the v2 weights.
|
| 5 |
+
|
| 6 |
This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
|
| 7 |
**Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
|
| 8 |
leaderboard score. The only comparisons in this report are: the untuned
|
|
|
|
| 283 |
|
| 284 |
## Training data provenance
|
| 285 |
|
| 286 |
+
The following cohort families make up the 46-cohort, 520,754-row training mixture used by
|
| 287 |
+
both Standard One 8B and Standard One 3B. Apart from the public-dataset items listed below, every row is synthetic (deterministic generators with visible checkers; a subset was teacher-reviewed) or format-augmented. No JevBench evaluation scenario appears in any
|
|
|
|
| 288 |
training cohort.
|
| 289 |
|
| 290 |
| Cohort family | What it covers | Licence |
|
|
|
|
| 297 |
| Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
|
| 298 |
| Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
|
| 299 |
| Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
|
| 300 |
+
| Synthetic and format-augmented decision data (v2) | Decision items in varied layouts | Apache-2.0 (ours) |
|
| 301 |
+
| Public-dataset decision items | Decision items converted from public datasets (table below) | Each dataset's own licence (table below) |
|
| 302 |
| JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
|
| 303 |
|
| 304 |
+
Public datasets used (train splits where a dataset has one; CUAD: its single release file). Labels come from the datasets;
|
| 305 |
+
distractor options and derived labels are generated by code. Licences as stated by each dataset; attribution to the dataset authors.
|
| 306 |
+
|
| 307 |
+
| Dataset | Upstream id | Licence | Cohort |
|
| 308 |
+
|---|---|---|---|
|
| 309 |
+
| SQuAD 2.0 | `rajpurkar/squad_v2` | CC BY-SA 4.0 | `pub-abstain-v1` |
|
| 310 |
+
| ARC | `allenai/ai2_arc` | CC BY-SA 4.0 | `pub-breadth-v1` |
|
| 311 |
+
| BoolQ | `google/boolq` | CC BY-SA 3.0 | `pub-breadth-v1` |
|
| 312 |
+
| CommonsenseQA | `tau/commonsense_qa` | MIT | `pub-breadth-v1` |
|
| 313 |
+
| HellaSwag | `Rowan/hellaswag` | MIT | `pub-breadth-v1` |
|
| 314 |
+
| Banking77 | `PolyAI/banking77` | CC BY 4.0 | `pub-classify-v1` |
|
| 315 |
+
| Bias in Bios | `LabHC/bias_in_bios` | MIT | `pub-classify-v1` |
|
| 316 |
+
| Bitext customer support | `bitext/Bitext-customer-support-llm-chatbot-training-dataset` | CDLA-Sharing-1.0 | `pub-classify-v1` |
|
| 317 |
+
| CLINC150 | `clinc/clinc_oos` | CC BY 3.0 | `pub-classify-v1` |
|
| 318 |
+
| Amazon Counterfactual | `mteb/amazon_counterfactual` | CC BY 4.0 | `pub-classify-v1` |
|
| 319 |
+
| DBpedia-14 | `fancyzhx/dbpedia_14` | CC BY-SA 3.0 | `pub-classify-v1` |
|
| 320 |
+
| Dolly 15k | `databricks/databricks-dolly-15k` | CC BY-SA 3.0 | `pub-classify-v1`, `safety-judge-v1` |
|
| 321 |
+
| GoEmotions | `google-research-datasets/go_emotions` | Apache-2.0 | `pub-classify-v1` |
|
| 322 |
+
| MASSIVE | `AmazonScience/massive` | CC BY 4.0 | `pub-classify-v1` |
|
| 323 |
+
| Twitter Financial News Sentiment | `zeroshot/twitter-financial-news-sentiment` | MIT | `pub-classify-v1` |
|
| 324 |
+
| HelpSteer3 | `nvidia/HelpSteer3` | CC BY 4.0 | `pub-judge-v1` |
|
| 325 |
+
| HelpSteer2 | `nvidia/HelpSteer2` | CC BY 4.0 | `pub-judge-v1` |
|
| 326 |
+
| 2WikiMultihopQA | `framolfese/2WikiMultihopQA` | Apache-2.0 | `pub-multihop-v1` |
|
| 327 |
+
| HotpotQA | `hotpotqa/hotpot_qa` | CC BY-SA 4.0 | `pub-multihop-v1` |
|
| 328 |
+
| MuSiQue | `dgslibisey/MuSiQue` | CC BY 4.0 | `pub-multihop-v1` |
|
| 329 |
+
| QASC | `allenai/qasc` | CC BY 4.0 | `pub-multihop-v1` |
|
| 330 |
+
| DROP | `ucinlp/drop` | CC BY-SA 4.0 | `pub-numeric-v1` |
|
| 331 |
+
| GSM8K | `openai/gsm8k` | MIT | `pub-numeric-v1` |
|
| 332 |
+
| TempReason | `tonytan48/TempReason` | CC BY-SA 3.0 | `pub-numeric-v1` |
|
| 333 |
+
| MultiNLI | `nyu-mll/multi_nli` | OANC / CC BY-SA 3.0 / CC BY 3.0 | `pub-paraphrase-v1` |
|
| 334 |
+
| PAWS | `google-research-datasets/paws` | Google terms, free for any purpose | `pub-paraphrase-v1` |
|
| 335 |
+
| PAWS-X | `google-research-datasets/paws-x` | Google terms, free for any purpose | `pub-paraphrase-v1` |
|
| 336 |
+
| SNLI | `stanfordnlp/snli` | CC BY-SA 4.0 | `pub-paraphrase-v1` |
|
| 337 |
+
| WANLI | `alisawuffles/WANLI` | CC BY 4.0 | `pub-paraphrase-v1` |
|
| 338 |
+
| ContractNLI | `stanfordnlp.github.io/contract-nli` | CC BY 4.0 | `pub-rules-v1` |
|
| 339 |
+
| CUAD | `theatticusproject/cuad` | CC BY 4.0 | `pub-rules-v1` |
|
| 340 |
+
| ShARC | `sharc-data.github.io` | CC BY-SA 3.0 | `pub-rules-v1` |
|
| 341 |
+
| Jailbreak classification | `jackhhao/jailbreak-classification` | Apache-2.0 | `safety-judge-v1` |
|
| 342 |
+
| Prompt injections | `deepset/prompt-injections` | Apache-2.0 | `safety-judge-v1` |
|
| 343 |
+
| Aegis AI Content Safety 2.0 | `nvidia/Aegis-AI-Content-Safety-Dataset-2.0` | CC BY 4.0 | `content-safety-text-v1` |
|
| 344 |
+
| Jigsaw Toxic Comment Classification (mirror of the Kaggle data) | `tasksource/jigsaw_toxicity` | CC0 (data); comment text CC BY-SA 3.0 (Wikipedia) | `content-safety-text-v1` |
|
| 345 |
+
| Measuring Hate Speech | `ucberkeley-dlab/measuring-hate-speech` | CC BY 4.0 | `content-safety-text-v1` |
|
| 346 |
+
| Image safety classes | `deepghs/nsfw_detect` | MIT | `content-safety-image-v1`, `content-safety-image-2d-v1` |
|
| 347 |
+
|
| 348 |
## Public-overlap audit and disclosure
|
| 349 |
|
| 350 |
+
Run on the 46-cohort, 520,754-row training data that **both candidates train on**:
|
| 351 |
|
| 352 |
- **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
|
| 353 |
- **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
|
|
|
|
| 355 |
one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
|
| 356 |
response fully and correctly satisfy the request?"), reused as the complete instruction rather than
|
| 357 |
as a substring.
|
| 358 |
+
- 24 distinct shared word 8-grams (20,400 rows), all in known families: the rubric sentence above (also in `judge-realistic-train-v3`, `judge-multilingual-train-v1`, `judge-proxy-v1` and the format-augmented cohort), an insurance "cancellation takes effect" boilerplate clause, a "coverage reviewer" claims-adjudication phrase, and generic contract boilerplate (for example "controlled by or under common control with") in one public-dataset cohort.
|
| 359 |
+
- All other 44 cohorts: 0 exact state and 0 exact instruction matches.
|
|
|
|
|
|
|
|
|
|
| 360 |
|
| 361 |
+
**These 181 rows are kept and disclosed here, not regenerated.** They are 0.03 % of the mixture, and
|
| 362 |
the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
|
| 363 |
|
| 364 |
## Evidence file hashes
|
model-00001-of-00004.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4999724576
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5b29a2e3e86366c0b2e21c5038a16fa1f2bed8a2ca69e5e775608126570c89d4
|
| 3 |
size 4999724576
|
model-00002-of-00004.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4999820896
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:65c9000a4aafa0cb0978540cfdbb6017e39f14a208a8d0523d375656f5cf0343
|
| 3 |
size 4999820896
|
model-00003-of-00004.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4915917688
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe537b22d9b005873c25b0f8ca0548e24f6d3a4fb611db11becc6ba86c849a03
|
| 3 |
size 4915917688
|
model-00004-of-00004.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2920659992
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6c3f84685c69360d991bf02b81ac670fcb36cc1b9af2dd51e2137e0f3d36a24a
|
| 3 |
size 2920659992
|
release-manifest.json
CHANGED
|
@@ -1,24 +1,26 @@
|
|
| 1 |
{
|
| 2 |
"release": "StandardOne-8B",
|
|
|
|
| 3 |
"base_model": {
|
| 4 |
"repo": "mistralai/Ministral-3-8B-Instruct-2512-BF16",
|
| 5 |
"revision": "f6fae9795746f63c9be8344932f01275f3c63734"
|
| 6 |
},
|
| 7 |
"adapter": {
|
| 8 |
-
"source_run": "
|
| 9 |
-
"adapter_model_sha256": "
|
| 10 |
},
|
| 11 |
"merge_report_summary": {
|
| 12 |
"changed_params": 238,
|
| 13 |
"unchanged_params": 293,
|
| 14 |
"changed_outside_lm_projections": [],
|
| 15 |
-
"max_abs_delta": 0.
|
| 16 |
},
|
| 17 |
"serving": {
|
| 18 |
"wording": "native",
|
| 19 |
"system_prompt": null,
|
| 20 |
"temperature": 1.65,
|
| 21 |
"served_model_name": "standard-one-8b",
|
| 22 |
-
"engine": "SGLang 0.5.20"
|
|
|
|
| 23 |
}
|
| 24 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"release": "StandardOne-8B",
|
| 3 |
+
"version": "v2",
|
| 4 |
"base_model": {
|
| 5 |
"repo": "mistralai/Ministral-3-8B-Instruct-2512-BF16",
|
| 6 |
"revision": "f6fae9795746f63c9be8344932f01275f3c63734"
|
| 7 |
},
|
| 8 |
"adapter": {
|
| 9 |
+
"source_run": "r5-8b",
|
| 10 |
+
"adapter_model_sha256": "53e41238cd55567cfbc75efdf61d354da39771573f56f74bb1440c9ffdd1bd4a"
|
| 11 |
},
|
| 12 |
"merge_report_summary": {
|
| 13 |
"changed_params": 238,
|
| 14 |
"unchanged_params": 293,
|
| 15 |
"changed_outside_lm_projections": [],
|
| 16 |
+
"max_abs_delta": 0.00201416015625
|
| 17 |
},
|
| 18 |
"serving": {
|
| 19 |
"wording": "native",
|
| 20 |
"system_prompt": null,
|
| 21 |
"temperature": 1.65,
|
| 22 |
"served_model_name": "standard-one-8b",
|
| 23 |
+
"engine": "SGLang 0.5.20",
|
| 24 |
+
"note": "serving settings shown are those of v1.1; they are being re-fitted for v2"
|
| 25 |
}
|
| 26 |
}
|