File size: 18,177 Bytes
03223d7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
// Build the Experiment 1 report from the packaged choice-panel summary.
// Usage: node build_whitepaper.js path/to/output.docx
const fs = require("fs");
const path = require("path");
const {
  Document, Packer, Paragraph, TextRun, Table, TableRow, TableCell,
  WidthType, BorderStyle, AlignmentType, HeadingLevel, Footer, PageNumber,
} = require("docx");

const OUT = process.argv[2] || "OpenJEV_E4B_Experiment_1_Whitepaper.docx";
const EVAL = path.resolve(__dirname, "../../eval/choice_panel");
const summary = JSON.parse(fs.readFileSync(path.join(EVAL, "summary.json"), "utf8"));
const audit = JSON.parse(fs.readFileSync(path.join(EVAL, "leak_audit.json"), "utf8"));
const rowsHash = fs.readFileSync(path.join(EVAL, "rows.sha256"), "utf8").trim().split(/\s+/)[0];
const W = 9360;
const body = [];

function P(value, options = {}) {
  body.push(new Paragraph({
    children: [new TextRun({ text: value, bold: !!options.bold, size: options.size || 20 })],
    spacing: { after: options.after === undefined ? 135 : options.after, line: 285 },
    keepNext: !!options.keepNext,
  }));
}
function H1(value) { body.push(new Paragraph({ heading: HeadingLevel.HEADING_1, text: value })); }
function H2(value) { body.push(new Paragraph({ heading: HeadingLevel.HEADING_2, text: value })); }
function Bullets(values) {
  for (const value of values) body.push(new Paragraph({
    text: "• " + value, indent: { left: 320, hanging: 200 },
    spacing: { after: 65, line: 275 },
  }));
}
function T(headers, rows, widths, fontSize = 16) {
  const total = widths.reduce((a, b) => a + b, 0);
  const col = widths.map(x => Math.round(W * x / total));
  col[col.length - 1] += W - col.reduce((a, b) => a + b, 0);
  const border = { style: BorderStyle.SINGLE, color: "D9D9D9", size: 4 };
  const borders = { top: border, bottom: border, left: border, right: border };
  const cell = (value, i, header) => new TableCell({
    width: { size: col[i], type: WidthType.DXA }, borders,
    shading: { fill: header ? "DCE6F1" : "FFFFFF" },
    margins: { top: 90, bottom: 90, left: 95, right: 95 },
    children: String(value).split("\n").map(line => new Paragraph({
      children: [new TextRun({ text: line, bold: header, size: fontSize })],
      alignment: i === 0 ? AlignmentType.LEFT : AlignmentType.CENTER,
      spacing: { after: 0, line: 235 },
    })),
  });
  body.push(new Table({
    width: { size: W, type: WidthType.DXA }, columnWidths: col,
    rows: [
      new TableRow({ tableHeader: true, cantSplit: true, children: headers.map((h, i) => cell(h, i, true)) }),
      ...rows.map(r => new TableRow({ cantSplit: true, children: r.map((v, i) => cell(v, i, false)) })),
    ],
  }));
  P("", { after: 70 });
}
function pct(value, digits = 1) { return (100 * value).toFixed(digits) + "%"; }
function score(model, stage, task, expected) {
  const value = summary.tasks[model + "|" + stage + "|" + task];
  return value && value.rows === expected ? value : null;
}
function fmt(value) {
  return value ? pct(value.debiased_accuracy) + " [" + pct(value.debiased_ci95[0]) + ", " + pct(value.debiased_ci95[1]) + "]" : "In progress";
}
function delta(value) {
  return value ? (value.delta_pp >= 0 ? "+" : "") + value.delta_pp.toFixed(2) + " [" +
    value.ci95_pp[0].toFixed(2) + ", " + value.ci95_pp[1].toFixed(2) + "] pp" : "In progress";
}

const tasks = [
  ["GPQA Diamond", "gpqa_v2", 198, 4],
  ["Chess legal", "chess_legal_v2", 500, 4],
  ["GSM8K 4 choices", "gsm8k_mc4_v2", 1319, 4],
  ["GSM8K 10 choices", "gsm8k_mc10_v2", 1319, 10],
];
const COMPLETE = ["raw", "r7"].every(model => tasks.every(([, id, n]) =>
  score(model, "full", id, n) && score(model, "options_only", id, n)));

P("OpenJEV E4B 1.0", { bold: true, size: 46, after: 60 });
P("Training a direct-decision readout on Gemma 4 E4B", { size: 31, after: 120 });
P("Experiment 1 method, evaluation, and results", { size: 26, after: 320 });
P("27 September 2026 • Public release 1.0");
P("Technical report accompanying the bambamdevs/openjev-e4b model release. OpenJEV is independent research and is not affiliated with TypeSafe AI or the unrelated AlexWortega/openjev project.");

H1("Abstract");
P("OpenJEV E4B converts google/gemma-4-E4B-it into a closed-choice decision system. Given state, a question, and options, it returns a probability distribution from a direct head without generating text. Experiment 1 trained a LoRA adapter and hybrid decision head, then used exact-oracle Arena data, promotion gates, and selective decoder-layer updates. This report describes the checkpoint selected for public release 1.0.");
P("The choice-panel evaluation freezes shuffled GPQA rows, GSM8K options drawn from other problems, and a Chess legal-move task. All options are evaluated in cyclic order rotations. The main score chooses the largest probability after averaging over rotations; an options-only condition and construction audit test whether choice strings reveal the answer. On untouched Gemma, rotation-averaged accuracy is 32.8% on GPQA, 43.8% on Chess legal, 43.7% on GSM8K-4, and 20.5% on GSM8K-10. The same-answer rate across orders ranges from 0.6% to 10.1%. " +
  (COMPLETE ? "OpenJEV 1.0 reaches 53.8% on GSM8K-4 and 30.6% on GSM8K-10, gains of about 10.1 percentage points on each task. GPQA is unchanged at 32.8% and Chess legal is 42.6% versus Gemma's 43.8%. Its options-only GPQA score is 32.3%, limiting what the full GPQA result can show about use of the question. Paired comparisons follow." : "OpenJEV 1.0 scoring is in progress; this report shows only completed choice-panel results."));
P("Five standard public tasks use native option order. They give a separate comparison between the native Gemma LM-head readout and the OpenJEV decision head. The report records family exposure and precision differences and states what the evidence can support.");

H1("1 Research question and model");
P("The research question is whether a small open base model can serve as a typed decision component: choose among caller-provided options and expose probabilities that software can inspect. A request contains State, Question, Options, and a final Decision marker. The decision query uses the hidden state at the final marker; each option uses its final-token state and a span mean.");
P("The model extracts the text decoder from Gemma 4 E4B and attaches LoRA with rank 16, alpha 32, and zero dropout. Its hybrid head combines a normalized pointer score and a span-residual MLP, then applies softmax with released temperature 1.0. Gemma shares key/value states across layers 24–41, so the adapter and cumulative delta have no key/value projections in those layers. The public checkpoint consists of the pinned Gemma base, adapter, head, 65-tensor cumulative backbone delta, calibration, and packing code. Restoring only the adapter does not restore the model.");

H2("1.1 Model evolution and promotion");
P("Development began with a direct readout and LoRA proof of concept, expanded to a production training mixture and hard negatives, compared readout designs, and selected a proper-scoring hybrid head plus LoRA. The later Arena curriculum generated 24,000 exact-oracle examples across 20 decision families per internal training cycle and mixed them with replay. Early Arena training changed the head and LoRA; later cycles also trained selected decoder layers. Accepted base changes accumulate into the public checkpoint's 65 delta tensors across layers 37–41. Internal evaluation populations changed during development, so their percentages are not a controlled public comparison.");
P("A challenger could be promoted only when validation accuracy, schema accuracy, firewall accuracy, order consistency, and validation log loss stayed within recorded preservation margins. The promotion score weighted gains in Arena performance, order consistency, schema, firewall, validation accuracy, and validation NLL; acceptance required all checks and a score above 0.001. The selected public checkpoint's stored score is 0.024174. The gate record preserves the score and pass flags, but not every component value.");

H1("2 Evaluation protocol");
H2("2.1 Choice-panel tasks");
T(["Task", "Rows", "Options", "Row construction"], [
  ["GPQA Diamond", "198", "4", "Source answers shuffled with a fixed per-row seed"],
  ["GSM8K 4", "1,319", "4", "Gold plus distinct answers to other test problems of similar magnitude"],
  ["GSM8K 10", "1,319", "10", "Same construction with nine distractors"],
  ["Chess legal", "500", "4", "One legal plain-SAN move and three plain-SAN moves from other positions, illegal here"],
], [1.4, 0.7, 1.15, 6.15]);
P("Frozen rows are published as eval/choice_panel/rows.jsonl (SHA-256 " + rowsHash + "). GPQA comes from the 198-row Diamond mirror, GSM8K from the main test split, and Chess legal from seeded synthetic positions. Chess asks which move is legal in the position, not which is best. The GSM8K tasks are multiple-choice arithmetic, not open-ended solution generation.");
P("Every full-condition row is presented in all cyclic option rotations: four for GPQA, Chess legal, and GSM8K-4; ten for GSM8K-10. Probabilities are mapped to original option identities before averaging. The main accuracy applies argmax to that per-row mean. Mean per-order accuracy averages individual presentations. Same-answer rate measures whether the predicted option identity is unchanged in every rotation. Options-only replaces state and question with neutral text, using four rotations for four choices and five evenly spaced rotations for ten choices.");
P("The construction gate checks gold-position balance, fixed choice-string heuristics, and a cross-validated logistic scorer on choice-only features. The maximum heuristic score stays within its specified 95% chance band on all four tasks. That gate tests row construction; model options-only scores are reported separately.");
T(["Task", "Chance", "Gold-position range", "Highest construction heuristic", "Gate"],
  tasks.map(([name, id, , k]) => {
    const x = audit[id];
    return [name, pct(1 / k), pct(Math.min(...x.gold_position_share)) + "–" + pct(Math.max(...x.gold_position_share)), pct(x.max_heuristic), x.leak_gate];
  }), [2.1, 1, 2, 2.2, 0.9]);

H2("2.2 Systems and uncertainty");
P("Untouched Gemma uses native LM-head answer-letter scoring in NF4. OpenJEV 1.0 uses the frozen base in NF4, delta-touched decoder modules restored in dense FP16, its LoRA adapter, and an FP32 decision head. These are complete inference systems with different readouts. Wilson 95% intervals use rows as the unit of observation. Paired differences use identical rows, row bootstrap intervals, and exact McNemar tests; rotations of one row are not independent observations. Evaluation ran on a Windows RTX 3060 with local cached weights.");

H1("3 Choice-panel results");
T(["Task", "Gemma base [95% CI]", "OpenJEV 1.0 [95% CI]", "Gemma per order / same", "OpenJEV per order / same"],
  tasks.map(([name, id, n]) => {
    const raw = score("raw", "full", id, n), model = score("r7", "full", id, n);
    const order = x => x ? pct(x.rotation_mean_accuracy) + " / " + pct(x.order_consistency) : "In progress";
    return [name, fmt(raw), fmt(model), order(raw), order(model)];
  }), [1.6, 2.3, 2.3, 1.9, 1.9], 15);
P("The main score uses one averaged probability vector per row. The per-order and same-answer columns show option-order sensitivity. Same-row paired differences appear in Section 3.2 when both systems are complete.");

H2("3.1 Options-only control");
T(["Task", "Chance", "Gemma base [95% CI]", "OpenJEV 1.0 [95% CI]"],
  tasks.map(([name, id, n, k]) => [name, pct(1 / k), fmt(score("raw", "options_only", id, n)), fmt(score("r7", "options_only", id, n))]),
  [2.1, 1, 2.7, 2.7], 15);
P("Gemma's options-only scores are close to chance on GPQA and the two GSM8K tasks; Chess legal is 29.0%. OpenJEV 1.0 scores 32.3% on GPQA and 30.8% on Chess legal with only options, both above 25% chance. Its full GPQA score is 32.8%, so this result alone does not establish use of the question. OpenJEV's GSM8K options-only scores are near chance. The construction audit passes its gate, but model behavior reveals additional choice-only signal. This control does not explain individual full-condition decisions.");
if (COMPLETE) {
  H2("3.2 Paired comparisons");
  T(["Task", "OpenJEV 1.0 − Gemma [95% CI]", "McNemar p"], tasks.map(([name, id]) => {
    const pair = (summary.pairs || {})["raw->r7|full|" + id];
    return [name, delta(pair), pair ? pair.mcnemar_p.toPrecision(2) : "In progress"];
  }), [2.1, 5.1, 1.1], 15);
  P("Intervals are paired row-bootstrap intervals with 2,000 resamples. Exact McNemar tests use discordant row-level correct/incorrect outcomes after rotation averaging. The comparison measures complete decision systems.");
}

H1("4 Standard public tasks");
P("ARC-Easy, ARC-Challenge, WinoGrande, HellaSwag, and MMLU use the public-task runner and native option order. The table compares identical rows under that protocol. ARC, HellaSwag, and MMLU have training-family exposure. These scores should not be pooled with the rotation-averaged choice-panel scores as though they shared one protocol.");
T(["Task", "Gemma NF4", "OpenJEV 1.0", "Difference [95% CI]"], [
  ["ARC-Easy", "95.58%", "95.62%", "+0.04 [-0.80, +0.84] pp"],
  ["ARC-Challenge", "87.37%", "87.88%", "+0.51 [-1.37, +2.39] pp"],
  ["WinoGrande", "60.14%", "66.46%", "+6.31 [+3.79, +9.00] pp"],
  ["HellaSwag", "76.51%", "89.77%", "+13.26 [+12.40, +14.06] pp"],
  ["MMLU", "66.19%", "64.41%", "-1.77 [-2.58, -1.01] pp"],
], [2, 1.5, 1.7, 4]);
P("Under native order, OpenJEV and Gemma are similar on ARC-Easy and ARC-Challenge; OpenJEV is higher on WinoGrande and HellaSwag and lower on MMLU. HellaSwag and ARC have same-family training exposure. OpenJEV's WinoGrande NLL is 0.873, above the uniform binary reference ln 2 = 0.693; the released temperature is 1.0, and calibration does not transfer uniformly across tasks.");

H1("5 Interpretation and next controls");
P("Experiment 1 produces a working direct-decision model and an inspectable release record. The choice panel is the evidence for GPQA, closed-choice GSM8K, and Chess legal. Standard tasks provide separate end-to-end comparisons under native order. Current comparisons cannot attribute differences to the head, adapter, backbone updates, or inference precision separately.");
Bullets([
  ...(!COMPLETE ? ["Complete the OpenJEV 1.0 choice-panel run and analyze paired rows, options-only performance, and order consistency."] : []),
  "Score a frozen Gemma backbone with a trained decision head to separate readout effects from adapter and backbone changes.",
  "Match inference precision across controlled checkpoints before attributing a difference to selective backbone training.",
  "Use an open-ended GSM8K reference to characterize the cost of a closed-choice interface on arithmetic generation.",
  "Reserve a knowledge panel unused for model selection before making confirmatory preservation claims.",
]);

H1("Appendix A Release and reproduction record");
P("The base is google/gemma-4-E4B-it at revision ee0ef6023621cff504d758262d4e04895a5af4a2. The public 1.0 release maps to internal checkpoint v1.6 R7, with stored Arena score 0.024174. It has 65 cumulative delta tensors in decoder layers 37–41 and global temperature 1.0. The repository includes the adapter, head, delta, loader, calibration, row-level evaluation artifacts, and a SHA-256 manifest. Training code is not included.");
P("The choice-panel row source and scoring scripts are in eval/choice_panel. The standard-task harness is in eval/harness. Source-checkpoint and release-file hashes are in RELEASE_METADATA.json and MANIFEST.sha256. Training-family overlap notes are in docs/BENCHMARK_EXPOSURE.md. The restoration and scoring contract is in REPRODUCIBILITY.md.");
P("Valid restoration requires the pinned base, adapter, head, cumulative delta, calibration, and packer. NF4 evaluation runs the frozen base in 4-bit with FP16 compute and touched decoder modules in dense FP16. Dense BF16 inference is supported by the loader but has not been publicly benchmarked.");

H1("References");
P("Google. Gemma 4 E4B-it model card. https://huggingface.co/google/gemma-4-E4B-it");
P("Rein et al. GPQA: A Graduate-Level Google-Proof Q&A Benchmark. arXiv:2311.12022, 2023.");
P("Cobbe et al. Training Verifiers to Solve Math Word Problems. arXiv:2110.14168, 2021.");
P("McNemar. Note on the sampling error of the difference between correlated proportions or percentages. Psychometrika 12, 1947.");
P("Guo et al. On Calibration of Modern Neural Networks. ICML 2017.");

const doc = new Document({
  title: "OpenJEV E4B 1.0 Experiment 1 method, evaluation, and results",
  creator: "OpenJEV E4B",
  description: "OpenJEV E4B 1.0 Experiment 1 technical report",
  styles: {
    default: { document: { run: { font: "Calibri", size: 20, color: "000000" } } },
    paragraphStyles: [
      { id: "Heading1", name: "Heading 1", basedOn: "Normal", next: "Normal", quickFormat: true,
        run: { font: "Calibri", size: 29, bold: true, color: "000000" },
        paragraph: { spacing: { before: 300, after: 120 }, outlineLevel: 0 } },
      { id: "Heading2", name: "Heading 2", basedOn: "Normal", next: "Normal", quickFormat: true,
        run: { font: "Calibri", size: 24, bold: true, color: "000000" },
        paragraph: { spacing: { before: 220, after: 90 }, outlineLevel: 1 } },
    ],
  },
  sections: [{
    properties: { page: { size: { width: 12240, height: 15840 },
      margin: { top: 1260, right: 1440, bottom: 1260, left: 1440 } } },
    footers: { default: new Footer({ children: [new Paragraph({
      alignment: AlignmentType.CENTER,
      children: [new TextRun({ text: "OpenJEV E4B  •  Experiment 1  •  1.0  •  ", size: 16, color: "666666" }),
        new TextRun({ children: [PageNumber.CURRENT], size: 16, color: "666666" })],
    })] }) },
    children: body,
  }],
});

Packer.toBuffer(doc).then(buffer => {
  fs.mkdirSync(path.dirname(path.resolve(OUT)), { recursive: true });
  fs.writeFileSync(OUT, buffer);
  console.log("Wrote " + OUT + " (" + buffer.length + " bytes)");
});