MyeongHoJeong commited on
Commit
8418ae8
·
verified ·
1 Parent(s): 74e5248

Update weights

Browse files
MERGE_REPORT.json CHANGED
@@ -1,11 +1,11 @@
1
  {
2
  "snapshot": "mistralai/Ministral-3-8B-Instruct-2512-BF16@f6fae9795746f63c9be8344932f01275f3c63734",
3
- "adapter": "StandardOne-8B-LoRA (training run hard25-8b-r1)",
4
- "adapter_sha256": "9551923c162970ea8bc2c3a5faf3ea3762b48591c9f460064e531d09427fa41a",
5
  "changed_params": 238,
6
  "unchanged_params": 293,
7
  "changed_outside_lm_projections": [],
8
- "max_abs_delta": 0.001983642578125,
9
  "changed_sample": [
10
  [
11
  "model.language_model.layers.0.self_attn.q_proj.weight",
@@ -17,23 +17,23 @@
17
  ],
18
  [
19
  "model.language_model.layers.0.self_attn.v_proj.weight",
20
- 0.0010833740234375
21
  ],
22
  [
23
  "model.language_model.layers.0.self_attn.o_proj.weight",
24
- 0.001407623291015625
25
  ],
26
  [
27
  "model.language_model.layers.0.mlp.gate_proj.weight",
28
- 0.00113677978515625
29
  ],
30
  [
31
  "model.language_model.layers.0.mlp.up_proj.weight",
32
- 0.0011749267578125
33
  ],
34
  [
35
  "model.language_model.layers.0.mlp.down_proj.weight",
36
- 0.0005550384521484375
37
  ],
38
  [
39
  "model.language_model.layers.1.self_attn.q_proj.weight",
@@ -45,10 +45,10 @@
45
  "chat_template.jinja": "74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56",
46
  "config.json": "111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882",
47
  "generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
48
- "model-00001-of-00004.safetensors": "b5da315ae797d8f7c9add8bd08249968c5936df1cc3071f7fb34cfc0e91bc1f2",
49
- "model-00002-of-00004.safetensors": "4157651afa8d781ce6e4ba90c57bc4606e440a09ebe19ca0e14463b65f1a3076",
50
- "model-00003-of-00004.safetensors": "655d65b5bb646889bccd46166dcc31c1e4e183958dd3a5688d6434a3299d2c58",
51
- "model-00004-of-00004.safetensors": "a6c3828542f1e5fefdb0658131a6d06274b7abe11794e0d9747f451bf1525a34",
52
  "model.safetensors.index.json": "71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb",
53
  "params.json": "81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538",
54
  "processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
@@ -57,5 +57,5 @@
57
  "tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
58
  "tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
59
  },
60
- "seconds": 63.8
61
  }
 
1
  {
2
  "snapshot": "mistralai/Ministral-3-8B-Instruct-2512-BF16@f6fae9795746f63c9be8344932f01275f3c63734",
3
+ "adapter": "StandardOne-8B-LoRA v2 (training run r5-8b)",
4
+ "adapter_sha256": "53e41238cd55567cfbc75efdf61d354da39771573f56f74bb1440c9ffdd1bd4a",
5
  "changed_params": 238,
6
  "unchanged_params": 293,
7
  "changed_outside_lm_projections": [],
8
+ "max_abs_delta": 0.00201416015625,
9
  "changed_sample": [
10
  [
11
  "model.language_model.layers.0.self_attn.q_proj.weight",
 
17
  ],
18
  [
19
  "model.language_model.layers.0.self_attn.v_proj.weight",
20
+ 0.0011348724365234375
21
  ],
22
  [
23
  "model.language_model.layers.0.self_attn.o_proj.weight",
24
+ 0.001430511474609375
25
  ],
26
  [
27
  "model.language_model.layers.0.mlp.gate_proj.weight",
28
+ 0.00112152099609375
29
  ],
30
  [
31
  "model.language_model.layers.0.mlp.up_proj.weight",
32
+ 0.001220703125
33
  ],
34
  [
35
  "model.language_model.layers.0.mlp.down_proj.weight",
36
+ 0.00058746337890625
37
  ],
38
  [
39
  "model.language_model.layers.1.self_attn.q_proj.weight",
 
45
  "chat_template.jinja": "74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56",
46
  "config.json": "111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882",
47
  "generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
48
+ "model-00001-of-00004.safetensors": "5b29a2e3e86366c0b2e21c5038a16fa1f2bed8a2ca69e5e775608126570c89d4",
49
+ "model-00002-of-00004.safetensors": "65c9000a4aafa0cb0978540cfdbb6017e39f14a208a8d0523d375656f5cf0343",
50
+ "model-00003-of-00004.safetensors": "fe537b22d9b005873c25b0f8ca0548e24f6d3a4fb611db11becc6ba86c849a03",
51
+ "model-00004-of-00004.safetensors": "6c3f84685c69360d991bf02b81ac670fcb36cc1b9af2dd51e2137e0f3d36a24a",
52
  "model.safetensors.index.json": "71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb",
53
  "params.json": "81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538",
54
  "processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
 
57
  "tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
58
  "tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
59
  },
60
+ "seconds": 65.5
61
  }
README.md CHANGED
@@ -27,6 +27,12 @@ tags:
27
 
28
  # Standard One 8B
29
 
 
 
 
 
 
 
30
  Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
31
  `POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
32
  BF16 8B checkpoint and the server code.
@@ -192,8 +198,8 @@ A JevBench v1.4.1 run has been requested; the sealed-set result is not yet avail
192
  ## Model details
193
 
194
  - **Base model:** `mistralai/Ministral-3-8B-Instruct-2512-BF16`, revision `f6fae9795746f63c9be8344932f01275f3c63734` (Apache-2.0).
195
- - **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **44,564,480** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 214,559,872 bytes, sha256 `9551923c162970ea8bc2c3a5faf3ea3762b48591c9f460064e531d09427fa41a`.
196
- - **Merged BF16 checkpoint:** merging the adapter into the base changed **238 tensors** (293 unchanged), none outside the language-model projections, max absolute weight change **0.00198**.
197
  - **Serving details:** native chat-template wording, no system prompt, fixed temperature (T = 1.65, fitted on held-out calibration data); served model name `standard-one-8b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single caller-supplied option order, no rotation ensemble; 8,192-token context.
198
 
199
  | Path | Contents |
@@ -206,9 +212,13 @@ A JevBench v1.4.1 run has been requested; the sealed-set result is not yet avail
206
 
207
  ## Training data
208
 
209
- Training data spans document and field normalisation, judge/routing and answer-adequacy, stated-distribution probability, verification/scoring, adequacy-rubric style, abstention and robustness (paraphrase, trap, hard natural-language), and game- and Tetris board-state cohorts — every row is synthetic and solver-generated (deterministic generators with visible checkers; a subset was teacher-reviewed), licensed Apache-2.0 (ours); the JevBench public tiers used only for evaluation carry MIT. Full per-cohort breakdown (row counts, what each covers, licence): [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
 
 
 
 
210
 
211
- An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.05 % of the 382,576-row training mixture). These rows are kept and disclosed here rather than regenerated, since the overlap is limited to one rubric question's wording and never touches a scenario or an answer.
212
 
213
  ## Limitations
214
 
 
27
 
28
  # Standard One 8B
29
 
30
+ > **Updated weights (v2, 2026-09-26).** If you downloaded this model before, download it again or pin
31
+ > `revision="v2"`. Earlier versions stay available under the tags `v1` and `v1.1`.
32
+ > The benchmark figures and the serving temperature on this card are still those of v1.1 and are being updated for v2.
33
+
34
+ **Version:** v2
35
+
36
  Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
37
  `POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
38
  BF16 8B checkpoint and the server code.
 
198
  ## Model details
199
 
200
  - **Base model:** `mistralai/Ministral-3-8B-Instruct-2512-BF16`, revision `f6fae9795746f63c9be8344932f01275f3c63734` (Apache-2.0).
201
+ - **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **44,564,480** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 214,559,872 bytes, sha256 `53e41238cd55567cfbc75efdf61d354da39771573f56f74bb1440c9ffdd1bd4a`.
202
+ - **Merged BF16 checkpoint:** merging the adapter into the base changed **238 tensors** (293 unchanged), none outside the language-model projections, max absolute weight change **0.00201**.
203
  - **Serving details:** native chat-template wording, no system prompt, fixed temperature (T = 1.65, fitted on held-out calibration data); served model name `standard-one-8b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single caller-supplied option order, no rotation ensemble; 8,192-token context.
204
 
205
  | Path | Contents |
 
212
 
213
  ## Training data
214
 
215
+ Training data is synthetic and format-augmented decision data plus decision items converted from public
216
+ datasets (listed below); the JevBench public tiers used only for evaluation carry MIT. Full per-cohort breakdown (row counts, what each covers, licence): [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
217
+
218
+ Public datasets used (train splits where the dataset has one; licence as stated by each dataset; labels come from the
219
+ datasets, distractor options are generated by code): SQuAD 2.0 (CC BY-SA 4.0), ARC (CC BY-SA 4.0), BoolQ (CC BY-SA 3.0), CommonsenseQA (MIT), HellaSwag (MIT), Banking77 (CC BY 4.0), Bias in Bios (MIT), Bitext customer support (CDLA-Sharing-1.0), CLINC150 (CC BY 3.0), Amazon Counterfactual (CC BY 4.0), DBpedia-14 (CC BY-SA 3.0), Dolly 15k (CC BY-SA 3.0), GoEmotions (Apache-2.0), MASSIVE (CC BY 4.0), Twitter Financial News Sentiment (MIT), HelpSteer3 (CC BY 4.0), HelpSteer2 (CC BY 4.0), 2WikiMultihopQA (Apache-2.0), HotpotQA (CC BY-SA 4.0), MuSiQue (CC BY 4.0), QASC (CC BY 4.0), DROP (CC BY-SA 4.0), GSM8K (MIT), TempReason (CC BY-SA 3.0), MultiNLI (OANC / CC BY-SA 3.0 / CC BY 3.0), PAWS (Google terms, free for any purpose), PAWS-X (Google terms, free for any purpose), SNLI (CC BY-SA 4.0), WANLI (CC BY 4.0), ContractNLI (CC BY 4.0), CUAD (CC BY 4.0), ShARC (CC BY-SA 3.0), Jailbreak classification (Apache-2.0), Prompt injections (Apache-2.0), Aegis AI Content Safety 2.0 (CC BY 4.0), Jigsaw Toxic Comment Classification (mirror of the Kaggle data) (CC0 (data); comment text CC BY-SA 3.0 (Wikipedia)), Measuring Hate Speech (CC BY 4.0), Image safety classes (MIT). Upstream ids and the cohort each one feeds: [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
220
 
221
+ An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.03 % of the 520,754-row training mixture). These rows are kept and disclosed here rather than regenerated, since the overlap is limited to one rubric question's wording and never touches a scenario or an answer.
222
 
223
  ## Limitations
224
 
SHA256SUMS CHANGED
@@ -1,12 +1,12 @@
1
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
2
- 1f39839e969d874d397d3842eb702f39516ec784f546ce18220e9836bffb2793 MERGE_REPORT.json
3
  20486cf6b2e42440b0aea913ea475c818664f34e76e4501651d6cb21987a508b NOTICE
4
  de8701f5a28091cdbc601d0625769a1356c6e31962df4fc2c2a559fe4a67369f QUICKSTART.md
5
- f63164a2215549b8675db36f0e901e4471d4fd333b09dcdca47975ad82aee552 README.md
6
  66e0fda8c8ab269ccd010dca6196931d475a22c3f2eebe5b6b3a2ec33a354d0e SYSTEM_PROMPT.txt
7
  74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56 chat_template.jinja
8
  111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882 config.json
9
- 868cb61534ddc6f57b8718113c54391b2d7fae46d9b72b6ae59b3c96d4994d2c docs/BENCHMARKS.md
10
  96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
11
  28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
12
  7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
@@ -25,14 +25,14 @@ c0eb0fde38c44b1f3f62f6a79cc45c3e89cad1f782d1c6cc40526f69e8595ff6 evidence/serve
25
  d47f43be5bd7767b80ed27a96a9a85ece1495b37fb9365e575b1df6b32fbc509 evidence/served-nosys-latency-idle-gpu.json
26
  b87e0a9a48ebb9fa933fe0db470b506cbe1850824abf65dfebf6ad6c5f58cdda evidence/served-nosys-original-report.json
27
  e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
28
- b5da315ae797d8f7c9add8bd08249968c5936df1cc3071f7fb34cfc0e91bc1f2 model-00001-of-00004.safetensors
29
- 4157651afa8d781ce6e4ba90c57bc4606e440a09ebe19ca0e14463b65f1a3076 model-00002-of-00004.safetensors
30
- 655d65b5bb646889bccd46166dcc31c1e4e183958dd3a5688d6434a3299d2c58 model-00003-of-00004.safetensors
31
- a6c3828542f1e5fefdb0658131a6d06274b7abe11794e0d9747f451bf1525a34 model-00004-of-00004.safetensors
32
  71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb model.safetensors.index.json
33
  81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538 params.json
34
  ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
35
- ad763bb1baa96641da64f113a8fdf28ec7ed798d081faa16749d81336b6efc68 release-manifest.json
36
  eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
37
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
38
  0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
 
1
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
2
+ 0fae9e8380c795cd6d6fa9f1733c408df7b27bf009d947ebb15adc96ad40dfab MERGE_REPORT.json
3
  20486cf6b2e42440b0aea913ea475c818664f34e76e4501651d6cb21987a508b NOTICE
4
  de8701f5a28091cdbc601d0625769a1356c6e31962df4fc2c2a559fe4a67369f QUICKSTART.md
5
+ ebcf3db9b80fb72f46b2616ed24ac37f10610eb7173f98f6247054d76da48bac README.md
6
  66e0fda8c8ab269ccd010dca6196931d475a22c3f2eebe5b6b3a2ec33a354d0e SYSTEM_PROMPT.txt
7
  74eeb55fd3341286ec3fd44e902b7120721acc81cd394e96b431f85e93a1ea56 chat_template.jinja
8
  111827aa7ae6d936f48afb2f0776f51b00add45c857c708af2b893499f76d882 config.json
9
+ 141067bb117600df9e09a46098e6a1f26e685ea5eab77f73b20352bce6d503d6 docs/BENCHMARKS.md
10
  96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
11
  28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
12
  7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
 
25
  d47f43be5bd7767b80ed27a96a9a85ece1495b37fb9365e575b1df6b32fbc509 evidence/served-nosys-latency-idle-gpu.json
26
  b87e0a9a48ebb9fa933fe0db470b506cbe1850824abf65dfebf6ad6c5f58cdda evidence/served-nosys-original-report.json
27
  e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
28
+ 5b29a2e3e86366c0b2e21c5038a16fa1f2bed8a2ca69e5e775608126570c89d4 model-00001-of-00004.safetensors
29
+ 65c9000a4aafa0cb0978540cfdbb6017e39f14a208a8d0523d375656f5cf0343 model-00002-of-00004.safetensors
30
+ fe537b22d9b005873c25b0f8ca0548e24f6d3a4fb611db11becc6ba86c849a03 model-00003-of-00004.safetensors
31
+ 6c3f84685c69360d991bf02b81ac670fcb36cc1b9af2dd51e2137e0f3d36a24a model-00004-of-00004.safetensors
32
  71e022361d37d84c2b6224acbddca2c74bedcc7addc523cfc6a57420d4b22ceb model.safetensors.index.json
33
  81b8377f36c5b3d60900b333214d85160a0500b84576e969092c6fd214f69538 params.json
34
  ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
35
+ 2eb900bdf73188c24ad9e8c0d1f47aadf0b6531eb33a0ce8929b52914cc183b3 release-manifest.json
36
  eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
37
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
38
  0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
docs/BENCHMARKS.md CHANGED
@@ -1,5 +1,8 @@
1
  # Standard One — full benchmark report
2
 
 
 
 
3
  This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
4
  **Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
5
  leaderboard score. The only comparisons in this report are: the untuned
@@ -280,9 +283,8 @@ and in `release-manifest.json`, identical for both candidates.
280
 
281
  ## Training data provenance
282
 
283
- The following cohort families make up the identical 34-cohort, 382,576-row training mixture used by
284
- both Standard One 8B and Standard One 3B. Every row is synthetic and solver-generated (deterministic generators with
285
- visible checkers; a subset was teacher-reviewed). No JevBench evaluation scenario appears in any
286
  training cohort.
287
 
288
  | Cohort family | What it covers | Licence |
@@ -295,11 +297,57 @@ training cohort.
295
  | Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
296
  | Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
297
  | Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
 
 
298
  | JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
299
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
300
  ## Public-overlap audit and disclosure
301
 
302
- Run on the exact final 34-cohort, 382,576-row mixture that **both candidates train on**:
303
 
304
  - **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
305
  - **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
@@ -307,13 +355,10 @@ Run on the exact final 34-cohort, 382,576-row mixture that **both candidates tra
307
  one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
308
  response fully and correctly satisfy the request?"), reused as the complete instruction rather than
309
  as a substring.
310
- - 16 distinct shared word 8-grams (12 distinct phrases, 9,756 rows), all in three known families: the
311
- rubric sentence above (also in `judge-realistic-train-v3`, `judge-multilingual-train-v1`,
312
- `judge-proxy-v1`), an insurance "cancellation takes effect" boilerplate clause, and a "coverage
313
- reviewer" claims-adjudication phrase.
314
- - All other 32 cohorts: 0 exact state and 0 exact instruction matches.
315
 
316
- **These 181 rows are kept and disclosed here, not regenerated.** They are 0.05 % of the mixture, and
317
  the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
318
 
319
  ## Evidence file hashes
 
1
  # Standard One — full benchmark report
2
 
3
+ > **Note.** The benchmark figures in this report are for v1.1 and are being updated for v2. The training-data provenance and
4
+ > overlap-audit sections below already describe the v2 weights.
5
+
6
  This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
7
  **Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
8
  leaderboard score. The only comparisons in this report are: the untuned
 
283
 
284
  ## Training data provenance
285
 
286
+ The following cohort families make up the 46-cohort, 520,754-row training mixture used by
287
+ both Standard One 8B and Standard One 3B. Apart from the public-dataset items listed below, every row is synthetic (deterministic generators with visible checkers; a subset was teacher-reviewed) or format-augmented. No JevBench evaluation scenario appears in any
 
288
  training cohort.
289
 
290
  | Cohort family | What it covers | Licence |
 
297
  | Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
298
  | Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
299
  | Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
300
+ | Synthetic and format-augmented decision data (v2) | Decision items in varied layouts | Apache-2.0 (ours) |
301
+ | Public-dataset decision items | Decision items converted from public datasets (table below) | Each dataset's own licence (table below) |
302
  | JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
303
 
304
+ Public datasets used (train splits where a dataset has one; CUAD: its single release file). Labels come from the datasets;
305
+ distractor options and derived labels are generated by code. Licences as stated by each dataset; attribution to the dataset authors.
306
+
307
+ | Dataset | Upstream id | Licence | Cohort |
308
+ |---|---|---|---|
309
+ | SQuAD 2.0 | `rajpurkar/squad_v2` | CC BY-SA 4.0 | `pub-abstain-v1` |
310
+ | ARC | `allenai/ai2_arc` | CC BY-SA 4.0 | `pub-breadth-v1` |
311
+ | BoolQ | `google/boolq` | CC BY-SA 3.0 | `pub-breadth-v1` |
312
+ | CommonsenseQA | `tau/commonsense_qa` | MIT | `pub-breadth-v1` |
313
+ | HellaSwag | `Rowan/hellaswag` | MIT | `pub-breadth-v1` |
314
+ | Banking77 | `PolyAI/banking77` | CC BY 4.0 | `pub-classify-v1` |
315
+ | Bias in Bios | `LabHC/bias_in_bios` | MIT | `pub-classify-v1` |
316
+ | Bitext customer support | `bitext/Bitext-customer-support-llm-chatbot-training-dataset` | CDLA-Sharing-1.0 | `pub-classify-v1` |
317
+ | CLINC150 | `clinc/clinc_oos` | CC BY 3.0 | `pub-classify-v1` |
318
+ | Amazon Counterfactual | `mteb/amazon_counterfactual` | CC BY 4.0 | `pub-classify-v1` |
319
+ | DBpedia-14 | `fancyzhx/dbpedia_14` | CC BY-SA 3.0 | `pub-classify-v1` |
320
+ | Dolly 15k | `databricks/databricks-dolly-15k` | CC BY-SA 3.0 | `pub-classify-v1`, `safety-judge-v1` |
321
+ | GoEmotions | `google-research-datasets/go_emotions` | Apache-2.0 | `pub-classify-v1` |
322
+ | MASSIVE | `AmazonScience/massive` | CC BY 4.0 | `pub-classify-v1` |
323
+ | Twitter Financial News Sentiment | `zeroshot/twitter-financial-news-sentiment` | MIT | `pub-classify-v1` |
324
+ | HelpSteer3 | `nvidia/HelpSteer3` | CC BY 4.0 | `pub-judge-v1` |
325
+ | HelpSteer2 | `nvidia/HelpSteer2` | CC BY 4.0 | `pub-judge-v1` |
326
+ | 2WikiMultihopQA | `framolfese/2WikiMultihopQA` | Apache-2.0 | `pub-multihop-v1` |
327
+ | HotpotQA | `hotpotqa/hotpot_qa` | CC BY-SA 4.0 | `pub-multihop-v1` |
328
+ | MuSiQue | `dgslibisey/MuSiQue` | CC BY 4.0 | `pub-multihop-v1` |
329
+ | QASC | `allenai/qasc` | CC BY 4.0 | `pub-multihop-v1` |
330
+ | DROP | `ucinlp/drop` | CC BY-SA 4.0 | `pub-numeric-v1` |
331
+ | GSM8K | `openai/gsm8k` | MIT | `pub-numeric-v1` |
332
+ | TempReason | `tonytan48/TempReason` | CC BY-SA 3.0 | `pub-numeric-v1` |
333
+ | MultiNLI | `nyu-mll/multi_nli` | OANC / CC BY-SA 3.0 / CC BY 3.0 | `pub-paraphrase-v1` |
334
+ | PAWS | `google-research-datasets/paws` | Google terms, free for any purpose | `pub-paraphrase-v1` |
335
+ | PAWS-X | `google-research-datasets/paws-x` | Google terms, free for any purpose | `pub-paraphrase-v1` |
336
+ | SNLI | `stanfordnlp/snli` | CC BY-SA 4.0 | `pub-paraphrase-v1` |
337
+ | WANLI | `alisawuffles/WANLI` | CC BY 4.0 | `pub-paraphrase-v1` |
338
+ | ContractNLI | `stanfordnlp.github.io/contract-nli` | CC BY 4.0 | `pub-rules-v1` |
339
+ | CUAD | `theatticusproject/cuad` | CC BY 4.0 | `pub-rules-v1` |
340
+ | ShARC | `sharc-data.github.io` | CC BY-SA 3.0 | `pub-rules-v1` |
341
+ | Jailbreak classification | `jackhhao/jailbreak-classification` | Apache-2.0 | `safety-judge-v1` |
342
+ | Prompt injections | `deepset/prompt-injections` | Apache-2.0 | `safety-judge-v1` |
343
+ | Aegis AI Content Safety 2.0 | `nvidia/Aegis-AI-Content-Safety-Dataset-2.0` | CC BY 4.0 | `content-safety-text-v1` |
344
+ | Jigsaw Toxic Comment Classification (mirror of the Kaggle data) | `tasksource/jigsaw_toxicity` | CC0 (data); comment text CC BY-SA 3.0 (Wikipedia) | `content-safety-text-v1` |
345
+ | Measuring Hate Speech | `ucberkeley-dlab/measuring-hate-speech` | CC BY 4.0 | `content-safety-text-v1` |
346
+ | Image safety classes | `deepghs/nsfw_detect` | MIT | `content-safety-image-v1`, `content-safety-image-2d-v1` |
347
+
348
  ## Public-overlap audit and disclosure
349
 
350
+ Run on the 46-cohort, 520,754-row training data that **both candidates train on**:
351
 
352
  - **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
353
  - **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
 
355
  one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
356
  response fully and correctly satisfy the request?"), reused as the complete instruction rather than
357
  as a substring.
358
+ - 24 distinct shared word 8-grams (20,400 rows), all in known families: the rubric sentence above (also in `judge-realistic-train-v3`, `judge-multilingual-train-v1`, `judge-proxy-v1` and the format-augmented cohort), an insurance "cancellation takes effect" boilerplate clause, a "coverage reviewer" claims-adjudication phrase, and generic contract boilerplate (for example "controlled by or under common control with") in one public-dataset cohort.
359
+ - All other 44 cohorts: 0 exact state and 0 exact instruction matches.
 
 
 
360
 
361
+ **These 181 rows are kept and disclosed here, not regenerated.** They are 0.03 % of the mixture, and
362
  the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
363
 
364
  ## Evidence file hashes
model-00001-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b5da315ae797d8f7c9add8bd08249968c5936df1cc3071f7fb34cfc0e91bc1f2
3
  size 4999724576
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5b29a2e3e86366c0b2e21c5038a16fa1f2bed8a2ca69e5e775608126570c89d4
3
  size 4999724576
model-00002-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4157651afa8d781ce6e4ba90c57bc4606e440a09ebe19ca0e14463b65f1a3076
3
  size 4999820896
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:65c9000a4aafa0cb0978540cfdbb6017e39f14a208a8d0523d375656f5cf0343
3
  size 4999820896
model-00003-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:655d65b5bb646889bccd46166dcc31c1e4e183958dd3a5688d6434a3299d2c58
3
  size 4915917688
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fe537b22d9b005873c25b0f8ca0548e24f6d3a4fb611db11becc6ba86c849a03
3
  size 4915917688
model-00004-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a6c3828542f1e5fefdb0658131a6d06274b7abe11794e0d9747f451bf1525a34
3
  size 2920659992
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6c3f84685c69360d991bf02b81ac670fcb36cc1b9af2dd51e2137e0f3d36a24a
3
  size 2920659992
release-manifest.json CHANGED
@@ -1,24 +1,26 @@
1
  {
2
  "release": "StandardOne-8B",
 
3
  "base_model": {
4
  "repo": "mistralai/Ministral-3-8B-Instruct-2512-BF16",
5
  "revision": "f6fae9795746f63c9be8344932f01275f3c63734"
6
  },
7
  "adapter": {
8
- "source_run": "hard25-8b-r1",
9
- "adapter_model_sha256": "9551923c162970ea8bc2c3a5faf3ea3762b48591c9f460064e531d09427fa41a"
10
  },
11
  "merge_report_summary": {
12
  "changed_params": 238,
13
  "unchanged_params": 293,
14
  "changed_outside_lm_projections": [],
15
- "max_abs_delta": 0.001983642578125
16
  },
17
  "serving": {
18
  "wording": "native",
19
  "system_prompt": null,
20
  "temperature": 1.65,
21
  "served_model_name": "standard-one-8b",
22
- "engine": "SGLang 0.5.20"
 
23
  }
24
  }
 
1
  {
2
  "release": "StandardOne-8B",
3
+ "version": "v2",
4
  "base_model": {
5
  "repo": "mistralai/Ministral-3-8B-Instruct-2512-BF16",
6
  "revision": "f6fae9795746f63c9be8344932f01275f3c63734"
7
  },
8
  "adapter": {
9
+ "source_run": "r5-8b",
10
+ "adapter_model_sha256": "53e41238cd55567cfbc75efdf61d354da39771573f56f74bb1440c9ffdd1bd4a"
11
  },
12
  "merge_report_summary": {
13
  "changed_params": 238,
14
  "unchanged_params": 293,
15
  "changed_outside_lm_projections": [],
16
+ "max_abs_delta": 0.00201416015625
17
  },
18
  "serving": {
19
  "wording": "native",
20
  "system_prompt": null,
21
  "temperature": 1.65,
22
  "served_model_name": "standard-one-8b",
23
+ "engine": "SGLang 0.5.20",
24
+ "note": "serving settings shown are those of v1.1; they are being re-fitted for v2"
25
  }
26
  }