MyeongHoJeong commited on
Commit
16767f1
·
verified ·
1 Parent(s): e351424

Update weights

Browse files
MERGE_REPORT.json CHANGED
@@ -1,15 +1,15 @@
1
  {
2
  "snapshot": "mistralai/Ministral-3-3B-Instruct-2512-BF16@b6d637bef2393152b3da2b2fde72eecdee30557e",
3
- "adapter": "StandardOne-3B-LoRA (training run hard25-3b-r1)",
4
- "adapter_sha256": "7bc8bfff4be52e0c9c6f605faf1eebc5e8a966ce103793c7f6970e29056ef5d3",
5
  "changed_params": 182,
6
  "unchanged_params": 276,
7
  "changed_outside_lm_projections": [],
8
- "max_abs_delta": 0.0022978782653808594,
9
  "changed_sample": [
10
  [
11
  "model.language_model.layers.0.self_attn.q_proj.weight",
12
- 0.0011186599731445312
13
  ],
14
  [
15
  "model.language_model.layers.0.self_attn.k_proj.weight",
@@ -17,19 +17,19 @@
17
  ],
18
  [
19
  "model.language_model.layers.0.self_attn.v_proj.weight",
20
- 0.0014190673828125
21
  ],
22
  [
23
  "model.language_model.layers.0.self_attn.o_proj.weight",
24
- 0.001007080078125
25
  ],
26
  [
27
  "model.language_model.layers.0.mlp.gate_proj.weight",
28
- 0.00202178955078125
29
  ],
30
  [
31
  "model.language_model.layers.0.mlp.up_proj.weight",
32
- 0.0017843246459960938
33
  ],
34
  [
35
  "model.language_model.layers.0.mlp.down_proj.weight",
@@ -37,7 +37,7 @@
37
  ],
38
  [
39
  "model.language_model.layers.1.self_attn.q_proj.weight",
40
- 0.001434326171875
41
  ]
42
  ],
43
  "file_sha256": {
@@ -45,8 +45,8 @@
45
  "chat_template.jinja": "0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b",
46
  "config.json": "c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3",
47
  "generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
48
- "model-00001-of-00002.safetensors": "8a02fdf56e2c3fdf75b303a1c617ed8bebe06d7e934e7ef56af3689a7a2552a5",
49
- "model-00002-of-00002.safetensors": "1d8099875a6e4466dace9ceab20c38ffee02903dbb2acee9d5b0565b6e64da2a",
50
  "model.safetensors.index.json": "55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e",
51
  "params.json": "d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59",
52
  "processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
@@ -55,5 +55,5 @@
55
  "tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
56
  "tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
57
  },
58
- "seconds": 26.7
59
  }
 
1
  {
2
  "snapshot": "mistralai/Ministral-3-3B-Instruct-2512-BF16@b6d637bef2393152b3da2b2fde72eecdee30557e",
3
+ "adapter": "StandardOne-3B-LoRA v2 (training run r5-3b)",
4
+ "adapter_sha256": "a8e3eb341e27c1a49a282e327c2a2038906abb0eee771e80debbb4a4c45f7afc",
5
  "changed_params": 182,
6
  "unchanged_params": 276,
7
  "changed_outside_lm_projections": [],
8
+ "max_abs_delta": 0.002337932586669922,
9
  "changed_sample": [
10
  [
11
  "model.language_model.layers.0.self_attn.q_proj.weight",
12
+ 0.0010986328125
13
  ],
14
  [
15
  "model.language_model.layers.0.self_attn.k_proj.weight",
 
17
  ],
18
  [
19
  "model.language_model.layers.0.self_attn.v_proj.weight",
20
+ 0.001434326171875
21
  ],
22
  [
23
  "model.language_model.layers.0.self_attn.o_proj.weight",
24
+ 0.00103759765625
25
  ],
26
  [
27
  "model.language_model.layers.0.mlp.gate_proj.weight",
28
+ 0.00200653076171875
29
  ],
30
  [
31
  "model.language_model.layers.0.mlp.up_proj.weight",
32
+ 0.00177001953125
33
  ],
34
  [
35
  "model.language_model.layers.0.mlp.down_proj.weight",
 
37
  ],
38
  [
39
  "model.language_model.layers.1.self_attn.q_proj.weight",
40
+ 0.00146484375
41
  ]
42
  ],
43
  "file_sha256": {
 
45
  "chat_template.jinja": "0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b",
46
  "config.json": "c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3",
47
  "generation_config.json": "e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677",
48
+ "model-00001-of-00002.safetensors": "7029a24151316b12d84b0e666c6af7d46bfe07e9549d59e513bb34afabb43073",
49
+ "model-00002-of-00002.safetensors": "ad184b56624eaca81a22c7fb7bef034edd25bd435cfb173a76d0db1b2c67ff2b",
50
  "model.safetensors.index.json": "55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e",
51
  "params.json": "d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59",
52
  "processor_config.json": "ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002",
 
55
  "tokenizer.json": "d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135",
56
  "tokenizer_config.json": "f59f7294e4f26383d0ea93840fe21cf197784be0842a8301a0343e8c34ed0d6d"
57
  },
58
+ "seconds": 25.9
59
  }
README.md CHANGED
@@ -27,6 +27,12 @@ tags:
27
 
28
  # Standard One 3B
29
 
 
 
 
 
 
 
30
  Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
31
  `POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
32
  BF16 3B checkpoint; the server code is in [StandardOne-8B](https://huggingface.co/StandardThinking/StandardOne-8B).
@@ -192,7 +198,7 @@ A JevBench v1.4.1 run has been requested; the sealed-set result is not yet avail
192
  ## Model details
193
 
194
  - **Base model:** `mistralai/Ministral-3-3B-Instruct-2512-BF16`, revision `b6d637bef2393152b3da2b2fde72eecdee30557e` (Apache-2.0).
195
- - **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **24,707,072** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 135,113,048 bytes, sha256 `7bc8bfff4be52e0c9c6f605faf1eebc5e8a966ce103793c7f6970e29056ef5d3`.
196
  - **Merged BF16 checkpoint:** merging the adapter into the base changed **182 tensors**, none of them outside the language-model projections, maximum absolute weight change **0.0023**.
197
  - **Serving details:** native chat-template wording, served without a system prompt, at a fixed temperature (T = 1.55, fitted on held-out calibration data); served model name `standard-one-3b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single, caller-supplied option order, no rotation ensemble; 8,192-token context.
198
 
@@ -207,9 +213,13 @@ The server code and full quick-start guide live in `StandardThinking/StandardOne
207
 
208
  ## Training data
209
 
210
- Trains on the identical mixture as `StandardThinking/StandardOne-8B` — same cohorts, same row counts, same overlap-audit result, not a reduced subset. Training data spans document and field normalisation, judge/routing and answer-adequacy, stated-distribution probability, verification/scoring, adequacy-rubric style, abstention and robustness (paraphrase, trap, hard natural-language), and game- and Tetris board-state cohorts — every row is synthetic and solver-generated (deterministic generators with visible checkers; a subset was teacher-reviewed), licensed Apache-2.0 (ours); the JevBench public tiers used only for evaluation carry MIT. Full per-cohort breakdown: [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
 
 
 
 
211
 
212
- An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.05 % of the 382,576-row training mixture). These rows are kept and disclosed here rather than regenerated, since the overlap is limited to one rubric question's wording and never touches a scenario or an answer.
213
 
214
  ## Limitations
215
 
 
27
 
28
  # Standard One 3B
29
 
30
+ > **Updated weights (v2, 2026-09-26).** If you downloaded this model before, download it again or pin
31
+ > `revision="v2"`. Earlier versions stay available under the tags `v1` and `v1.1`.
32
+ > The benchmark figures and the serving temperature on this card are still those of v1.1 and are being updated for v2.
33
+
34
+ **Version:** v2
35
+
36
  Standard One scores a bounded set of answers for a supplied scenario and returns probabilities through
37
  `POST /v1/systemone`. It does not generate free-form response text. This repository contains the merged
38
  BF16 3B checkpoint; the server code is in [StandardOne-8B](https://huggingface.co/StandardThinking/StandardOne-8B).
 
198
  ## Model details
199
 
200
  - **Base model:** `mistralai/Ministral-3-3B-Instruct-2512-BF16`, revision `b6d637bef2393152b3da2b2fde72eecdee30557e` (Apache-2.0).
201
+ - **Adapter:** LoRA r=16, α=32, dropout 0, on `q_proj k_proj v_proj o_proj gate_proj up_proj down_proj` of the language-model projections only (vision tower and multimodal projector excluded), **24,707,072** trainable parameters, PEFT 0.21.0. Adapter file `adapter_model.safetensors`, 135,113,048 bytes, sha256 `a8e3eb341e27c1a49a282e327c2a2038906abb0eee771e80debbb4a4c45f7afc`.
202
  - **Merged BF16 checkpoint:** merging the adapter into the base changed **182 tensors**, none of them outside the language-model projections, maximum absolute weight change **0.0023**.
203
  - **Serving details:** native chat-template wording, served without a system prompt, at a fixed temperature (T = 1.55, fitted on held-out calibration data); served model name `standard-one-3b` behind stock SGLang 0.5.20 via `jev-adapter` (`POST /v1/systemone`); single, caller-supplied option order, no rotation ensemble; 8,192-token context.
204
 
 
213
 
214
  ## Training data
215
 
216
+ Trains on the same data sources as `StandardThinking/StandardOne-8B`, not a reduced subset. Training data is synthetic and format-augmented decision data plus decision items converted from public
217
+ datasets (listed below); the JevBench public tiers used only for evaluation carry MIT. Full per-cohort breakdown: [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
218
+
219
+ Public datasets used (train splits where the dataset has one; licence as stated by each dataset; labels come from the
220
+ datasets, distractor options are generated by code): SQuAD 2.0 (CC BY-SA 4.0), ARC (CC BY-SA 4.0), BoolQ (CC BY-SA 3.0), CommonsenseQA (MIT), HellaSwag (MIT), Banking77 (CC BY 4.0), Bias in Bios (MIT), Bitext customer support (CDLA-Sharing-1.0), CLINC150 (CC BY 3.0), Amazon Counterfactual (CC BY 4.0), DBpedia-14 (CC BY-SA 3.0), Dolly 15k (CC BY-SA 3.0), GoEmotions (Apache-2.0), MASSIVE (CC BY 4.0), Twitter Financial News Sentiment (MIT), HelpSteer3 (CC BY 4.0), HelpSteer2 (CC BY 4.0), 2WikiMultihopQA (Apache-2.0), HotpotQA (CC BY-SA 4.0), MuSiQue (CC BY 4.0), QASC (CC BY 4.0), DROP (CC BY-SA 4.0), GSM8K (MIT), TempReason (CC BY-SA 3.0), MultiNLI (OANC / CC BY-SA 3.0 / CC BY 3.0), PAWS (Google terms, free for any purpose), PAWS-X (Google terms, free for any purpose), SNLI (CC BY-SA 4.0), WANLI (CC BY 4.0), ContractNLI (CC BY 4.0), CUAD (CC BY 4.0), ShARC (CC BY-SA 3.0), Jailbreak classification (Apache-2.0), Prompt injections (Apache-2.0), Aegis AI Content Safety 2.0 (CC BY 4.0), Jigsaw Toxic Comment Classification (mirror of the Kaggle data) (CC0 (data); comment text CC BY-SA 3.0 (Wikipedia)), Measuring Hate Speech (CC BY 4.0), Image safety classes (MIT). Upstream ids and the cohort each one feeds: [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md#training-data-provenance).
221
 
222
+ An exact-text overlap audit against the public JevBench tiers found 0 exact scenario matches and 181 exact instruction matches — rows in two adequacy-rubric cohorts whose entire instruction field, a generic 58-character adequacy question, is byte-identical to one public hard-tier instruction (0.03 % of the 520,754-row training mixture). These rows are kept and disclosed here rather than regenerated, since the overlap is limited to one rubric question's wording and never touches a scenario or an answer.
223
 
224
  ## Limitations
225
 
SHA256SUMS CHANGED
@@ -1,12 +1,12 @@
1
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
2
- 6756de0e7de5d85d7f5b4ff2e5e75b4b74e2d2ad7f0addf5719e652ba53e15dc MERGE_REPORT.json
3
  f10343f951ecb3a72670e12720de6e04d59e9cb872e84b2ceddd7d54853bb7e7 NOTICE
4
  372d9160e95493d1d8e5c94d82b12fb0693fd36e4a50f5206d287ee74f189046 QUICKSTART.md
5
- e97c6a7f38db1a9f5b2a1fe8ed128b60d5d0c48e87671b6f2a108f155d60cce8 README.md
6
  331b249682cd52226c50e533f59825184997f3b31f9a62b2c3fea940db6999c5 SYSTEM_PROMPT.txt
7
  0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b chat_template.jinja
8
  c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3 config.json
9
- 868cb61534ddc6f57b8718113c54391b2d7fae46d9b72b6ae59b3c96d4994d2c docs/BENCHMARKS.md
10
  96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
11
  28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
12
  7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
@@ -25,12 +25,12 @@ dd888be78c0e32894cf04a8e3fbf8757e7218873668fa4443ba47c7c1c48b6dc docs/assets/04
25
  7b9b1ee2f93166a817ead61bbc0f4c62984afe7d6c8affefd051007582d0a485 evidence/served-nosys-latency-idle-gpu.json
26
  5b217fe780b4c8c7cb292d57630b0f0a0bb5dcbe275f731272d6f0804cc14a68 evidence/served-nosys-original-report.json
27
  e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
28
- 8a02fdf56e2c3fdf75b303a1c617ed8bebe06d7e934e7ef56af3689a7a2552a5 model-00001-of-00002.safetensors
29
- 1d8099875a6e4466dace9ceab20c38ffee02903dbb2acee9d5b0565b6e64da2a model-00002-of-00002.safetensors
30
  55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e model.safetensors.index.json
31
  d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59 params.json
32
  ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
33
- fa228b3ddc6db30e10a55b8b7039646572f53e66f96c04a416f34480e9415e54 release-manifest.json
34
  eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
35
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
36
  0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
 
1
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 LICENSE
2
+ 916927708f20ec40252055f98ad7f09f5dcfacadc31e99d85bcac19f285b7bad MERGE_REPORT.json
3
  f10343f951ecb3a72670e12720de6e04d59e9cb872e84b2ceddd7d54853bb7e7 NOTICE
4
  372d9160e95493d1d8e5c94d82b12fb0693fd36e4a50f5206d287ee74f189046 QUICKSTART.md
5
+ faf66b200ab0a00f0e7f7c7174af1996c45376d3e529ed08ea7582b232c1e602 README.md
6
  331b249682cd52226c50e533f59825184997f3b31f9a62b2c3fea940db6999c5 SYSTEM_PROMPT.txt
7
  0701cfbdc2b7d44fdbad104dff604faee4b0543e8247624568777fe465746f9b chat_template.jinja
8
  c4446cf970fefd8b65f2e3119c30d406ea7026dca362dfb319c9aa08643131c3 config.json
9
+ 141067bb117600df9e09a46098e6a1f26e685ea5eab77f73b20352bce6d503d6 docs/BENCHMARKS.md
10
  96127670ad1e5bdd336a33118d6246330c27ac9e4fcb3f4f22767033809cd889 docs/assets/00-benchmark-card.png
11
  28c67a8c861f421fae5889af259d0850e7fb6eadb11f247adef72b94bf00b187 docs/assets/00-benchmark-card.svg
12
  7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2 docs/assets/02-latency-vs-qwen.png
 
25
  7b9b1ee2f93166a817ead61bbc0f4c62984afe7d6c8affefd051007582d0a485 evidence/served-nosys-latency-idle-gpu.json
26
  5b217fe780b4c8c7cb292d57630b0f0a0bb5dcbe275f731272d6f0804cc14a68 evidence/served-nosys-original-report.json
27
  e0923390059f84a9180b00e5501778acc45ea9856cd7f2fd68208b360927c677 generation_config.json
28
+ 7029a24151316b12d84b0e666c6af7d46bfe07e9549d59e513bb34afabb43073 model-00001-of-00002.safetensors
29
+ ad184b56624eaca81a22c7fb7bef034edd25bd435cfb173a76d0db1b2c67ff2b model-00002-of-00002.safetensors
30
  55641440eb385d8bc83a319b0a92f60fe45e6aef4b6992245eee91142237786e model.safetensors.index.json
31
  d15587c57574d2a92c66ff2552c6db5173d4c440ad47f7454dc51bad0ecaeb59 params.json
32
  ece2373c2ae391bce18785a1810543bc6173a1f28b6767cffab2c35dbea5f002 processor_config.json
33
+ f9913e0d49fc053df656058d13af2523b6263f9396efad7a6fa06e7a121e1acf release-manifest.json
34
  eef2a88a000d2ec294379755770c147327cb79d3d7278c711899bd36797d7a02 server/.gitignore
35
  1495e1e757ef4d0925a2350563cf5754bb23c51701a8ec4fb3c5cdcbedae6747 server/LICENSE
36
  0f1f023b916d38d7c12af3a71e09f5e82b819a7e156e4992118bb160ee63ede1 server/NOTICE
docs/BENCHMARKS.md CHANGED
@@ -1,5 +1,8 @@
1
  # Standard One — full benchmark report
2
 
 
 
 
3
  This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
4
  **Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
5
  leaderboard score. The only comparisons in this report are: the untuned
@@ -280,9 +283,8 @@ and in `release-manifest.json`, identical for both candidates.
280
 
281
  ## Training data provenance
282
 
283
- The following cohort families make up the identical 34-cohort, 382,576-row training mixture used by
284
- both Standard One 8B and Standard One 3B. Every row is synthetic and solver-generated (deterministic generators with
285
- visible checkers; a subset was teacher-reviewed). No JevBench evaluation scenario appears in any
286
  training cohort.
287
 
288
  | Cohort family | What it covers | Licence |
@@ -295,11 +297,57 @@ training cohort.
295
  | Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
296
  | Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
297
  | Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
 
 
298
  | JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
299
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
300
  ## Public-overlap audit and disclosure
301
 
302
- Run on the exact final 34-cohort, 382,576-row mixture that **both candidates train on**:
303
 
304
  - **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
305
  - **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
@@ -307,13 +355,10 @@ Run on the exact final 34-cohort, 382,576-row mixture that **both candidates tra
307
  one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
308
  response fully and correctly satisfy the request?"), reused as the complete instruction rather than
309
  as a substring.
310
- - 16 distinct shared word 8-grams (12 distinct phrases, 9,756 rows), all in three known families: the
311
- rubric sentence above (also in `judge-realistic-train-v3`, `judge-multilingual-train-v1`,
312
- `judge-proxy-v1`), an insurance "cancellation takes effect" boilerplate clause, and a "coverage
313
- reviewer" claims-adjudication phrase.
314
- - All other 32 cohorts: 0 exact state and 0 exact instruction matches.
315
 
316
- **These 181 rows are kept and disclosed here, not regenerated.** They are 0.05 % of the mixture, and
317
  the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
318
 
319
  ## Evidence file hashes
 
1
  # Standard One — full benchmark report
2
 
3
+ > **Note.** The benchmark figures in this report are for v1.1 and are being updated for v2. The training-data provenance and
4
+ > overlap-audit sections below already describe the v2 weights.
5
+
6
  This report covers both released systems, **Standard One 8B** (`StandardThinking/StandardOne-8B` / `-LoRA`) and
7
  **Standard One 3B** (`StandardThinking/StandardOne-3B` / `-LoRA`). It is our own measurement, not an official JevBench
8
  leaderboard score. The only comparisons in this report are: the untuned
 
283
 
284
  ## Training data provenance
285
 
286
+ The following cohort families make up the 46-cohort, 520,754-row training mixture used by
287
+ both Standard One 8B and Standard One 3B. Apart from the public-dataset items listed below, every row is synthetic (deterministic generators with visible checkers; a subset was teacher-reviewed) or format-augmented. No JevBench evaluation scenario appears in any
 
288
  training cohort.
289
 
290
  | Cohort family | What it covers | Licence |
 
297
  | Abstention & robustness (`abstain`, `paraphrase`, `trap`, `hard-natural`) | Abstaining when the state does not settle the question, paraphrased and adversarial wording, harder natural-language decisions | Apache-2.0 (ours) |
298
  | Game & board-state (`games-text-train`, `games-harness-train`, `games-harness-multilingual`) | Othello, chess, tic-tac-toe, snake and graph-problem decisions | Apache-2.0 (ours) |
299
  | Tetris board-state (`games-text-train-tetris`, `games-harness-tetris`, `tetris-drought-train`) | Tetris-specific board-state decisions | Apache-2.0 (ours) |
300
+ | Synthetic and format-augmented decision data (v2) | Decision items in varied layouts | Apache-2.0 (ours) |
301
+ | Public-dataset decision items | Decision items converted from public datasets (table below) | Each dataset's own licence (table below) |
302
  | JevBench public tiers (evaluation only, not training) | Easy / standard / hard held-out accuracy measurement | MIT |
303
 
304
+ Public datasets used (train splits where a dataset has one; CUAD: its single release file). Labels come from the datasets;
305
+ distractor options and derived labels are generated by code. Licences as stated by each dataset; attribution to the dataset authors.
306
+
307
+ | Dataset | Upstream id | Licence | Cohort |
308
+ |---|---|---|---|
309
+ | SQuAD 2.0 | `rajpurkar/squad_v2` | CC BY-SA 4.0 | `pub-abstain-v1` |
310
+ | ARC | `allenai/ai2_arc` | CC BY-SA 4.0 | `pub-breadth-v1` |
311
+ | BoolQ | `google/boolq` | CC BY-SA 3.0 | `pub-breadth-v1` |
312
+ | CommonsenseQA | `tau/commonsense_qa` | MIT | `pub-breadth-v1` |
313
+ | HellaSwag | `Rowan/hellaswag` | MIT | `pub-breadth-v1` |
314
+ | Banking77 | `PolyAI/banking77` | CC BY 4.0 | `pub-classify-v1` |
315
+ | Bias in Bios | `LabHC/bias_in_bios` | MIT | `pub-classify-v1` |
316
+ | Bitext customer support | `bitext/Bitext-customer-support-llm-chatbot-training-dataset` | CDLA-Sharing-1.0 | `pub-classify-v1` |
317
+ | CLINC150 | `clinc/clinc_oos` | CC BY 3.0 | `pub-classify-v1` |
318
+ | Amazon Counterfactual | `mteb/amazon_counterfactual` | CC BY 4.0 | `pub-classify-v1` |
319
+ | DBpedia-14 | `fancyzhx/dbpedia_14` | CC BY-SA 3.0 | `pub-classify-v1` |
320
+ | Dolly 15k | `databricks/databricks-dolly-15k` | CC BY-SA 3.0 | `pub-classify-v1`, `safety-judge-v1` |
321
+ | GoEmotions | `google-research-datasets/go_emotions` | Apache-2.0 | `pub-classify-v1` |
322
+ | MASSIVE | `AmazonScience/massive` | CC BY 4.0 | `pub-classify-v1` |
323
+ | Twitter Financial News Sentiment | `zeroshot/twitter-financial-news-sentiment` | MIT | `pub-classify-v1` |
324
+ | HelpSteer3 | `nvidia/HelpSteer3` | CC BY 4.0 | `pub-judge-v1` |
325
+ | HelpSteer2 | `nvidia/HelpSteer2` | CC BY 4.0 | `pub-judge-v1` |
326
+ | 2WikiMultihopQA | `framolfese/2WikiMultihopQA` | Apache-2.0 | `pub-multihop-v1` |
327
+ | HotpotQA | `hotpotqa/hotpot_qa` | CC BY-SA 4.0 | `pub-multihop-v1` |
328
+ | MuSiQue | `dgslibisey/MuSiQue` | CC BY 4.0 | `pub-multihop-v1` |
329
+ | QASC | `allenai/qasc` | CC BY 4.0 | `pub-multihop-v1` |
330
+ | DROP | `ucinlp/drop` | CC BY-SA 4.0 | `pub-numeric-v1` |
331
+ | GSM8K | `openai/gsm8k` | MIT | `pub-numeric-v1` |
332
+ | TempReason | `tonytan48/TempReason` | CC BY-SA 3.0 | `pub-numeric-v1` |
333
+ | MultiNLI | `nyu-mll/multi_nli` | OANC / CC BY-SA 3.0 / CC BY 3.0 | `pub-paraphrase-v1` |
334
+ | PAWS | `google-research-datasets/paws` | Google terms, free for any purpose | `pub-paraphrase-v1` |
335
+ | PAWS-X | `google-research-datasets/paws-x` | Google terms, free for any purpose | `pub-paraphrase-v1` |
336
+ | SNLI | `stanfordnlp/snli` | CC BY-SA 4.0 | `pub-paraphrase-v1` |
337
+ | WANLI | `alisawuffles/WANLI` | CC BY 4.0 | `pub-paraphrase-v1` |
338
+ | ContractNLI | `stanfordnlp.github.io/contract-nli` | CC BY 4.0 | `pub-rules-v1` |
339
+ | CUAD | `theatticusproject/cuad` | CC BY 4.0 | `pub-rules-v1` |
340
+ | ShARC | `sharc-data.github.io` | CC BY-SA 3.0 | `pub-rules-v1` |
341
+ | Jailbreak classification | `jackhhao/jailbreak-classification` | Apache-2.0 | `safety-judge-v1` |
342
+ | Prompt injections | `deepset/prompt-injections` | Apache-2.0 | `safety-judge-v1` |
343
+ | Aegis AI Content Safety 2.0 | `nvidia/Aegis-AI-Content-Safety-Dataset-2.0` | CC BY 4.0 | `content-safety-text-v1` |
344
+ | Jigsaw Toxic Comment Classification (mirror of the Kaggle data) | `tasksource/jigsaw_toxicity` | CC0 (data); comment text CC BY-SA 3.0 (Wikipedia) | `content-safety-text-v1` |
345
+ | Measuring Hate Speech | `ucberkeley-dlab/measuring-hate-speech` | CC BY 4.0 | `content-safety-text-v1` |
346
+ | Image safety classes | `deepghs/nsfw_detect` | MIT | `content-safety-image-v1`, `content-safety-image-2d-v1` |
347
+
348
  ## Public-overlap audit and disclosure
349
 
350
+ Run on the 46-cohort, 520,754-row training data that **both candidates train on**:
351
 
352
  - **0 exact state matches** — no training row's state/scenario duplicates any public JevBench item.
353
  - **181 exact instruction matches**: `hardstyle-train-v1` (87 of 3,900 rows) and `hardstyle-train-v2`
 
355
  one public `jevbench-hard` instruction — a 58-character generic adequacy-rubric question ("does the
356
  response fully and correctly satisfy the request?"), reused as the complete instruction rather than
357
  as a substring.
358
+ - 24 distinct shared word 8-grams (20,400 rows), all in known families: the rubric sentence above (also in `judge-realistic-train-v3`, `judge-multilingual-train-v1`, `judge-proxy-v1` and the format-augmented cohort), an insurance "cancellation takes effect" boilerplate clause, a "coverage reviewer" claims-adjudication phrase, and generic contract boilerplate (for example "controlled by or under common control with") in one public-dataset cohort.
359
+ - All other 44 cohorts: 0 exact state and 0 exact instruction matches.
 
 
 
360
 
361
+ **These 181 rows are kept and disclosed here, not regenerated.** They are 0.03 % of the mixture, and
362
  the exposure is limited to one hard-tier *question's* generic wording, never a *state* or an answer.
363
 
364
  ## Evidence file hashes
model-00001-of-00002.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8a02fdf56e2c3fdf75b303a1c617ed8bebe06d7e934e7ef56af3689a7a2552a5
3
  size 4995664864
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7029a24151316b12d84b0e666c6af7d46bfe07e9549d59e513bb34afabb43073
3
  size 4995664864
model-00002-of-00002.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1d8099875a6e4466dace9ceab20c38ffee02903dbb2acee9d5b0565b6e64da2a
3
  size 2702576200
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad184b56624eaca81a22c7fb7bef034edd25bd435cfb173a76d0db1b2c67ff2b
3
  size 2702576200
release-manifest.json CHANGED
@@ -1,24 +1,26 @@
1
  {
2
  "release": "StandardOne-3B",
 
3
  "base_model": {
4
  "repo": "mistralai/Ministral-3-3B-Instruct-2512-BF16",
5
  "revision": "b6d637bef2393152b3da2b2fde72eecdee30557e"
6
  },
7
  "adapter": {
8
- "source_run": "hard25-3b-r1",
9
- "adapter_model_sha256": "7bc8bfff4be52e0c9c6f605faf1eebc5e8a966ce103793c7f6970e29056ef5d3"
10
  },
11
  "merge_report_summary": {
12
  "changed_params": 182,
13
  "unchanged_params": 276,
14
  "changed_outside_lm_projections": [],
15
- "max_abs_delta": 0.0022978782653808594
16
  },
17
  "serving": {
18
  "wording": "native",
19
  "system_prompt": null,
20
  "temperature": 1.55,
21
  "served_model_name": "standard-one-3b",
22
- "engine": "SGLang 0.5.20"
 
23
  }
24
  }
 
1
  {
2
  "release": "StandardOne-3B",
3
+ "version": "v2",
4
  "base_model": {
5
  "repo": "mistralai/Ministral-3-3B-Instruct-2512-BF16",
6
  "revision": "b6d637bef2393152b3da2b2fde72eecdee30557e"
7
  },
8
  "adapter": {
9
+ "source_run": "r5-3b",
10
+ "adapter_model_sha256": "a8e3eb341e27c1a49a282e327c2a2038906abb0eee771e80debbb4a4c45f7afc"
11
  },
12
  "merge_report_summary": {
13
  "changed_params": 182,
14
  "unchanged_params": 276,
15
  "changed_outside_lm_projections": [],
16
+ "max_abs_delta": 0.002337932586669922
17
  },
18
  "serving": {
19
  "wording": "native",
20
  "system_prompt": null,
21
  "temperature": 1.55,
22
  "served_model_name": "standard-one-3b",
23
+ "engine": "SGLang 0.5.20",
24
+ "note": "serving settings shown are those of v1.1; they are being re-fitted for v2"
25
  }
26
  }