Spaces:
Running
Running
| <html lang="en"> | |
| <head> | |
| <meta charset="utf-8"> | |
| <meta name="viewport" content="width=device-width, initial-scale=1"> | |
| <meta name="description" content="Public, auditable MAT Nexus results on mechanically verifiable tasks."> | |
| <title>MAT Nexus — public results</title> | |
| <link rel="stylesheet" href="dashboard.css?v=4"> | |
| </head> | |
| <body> | |
| <a class="skip-link" href="#main" data-i18n="skip">Skip to results</a> | |
| <header class="hero"> | |
| <nav aria-label="Main navigation"> | |
| <a class="brand" href="../">MAT Nexus</a> | |
| <div class="nav-actions"> | |
| <div class="language-switch" role="group" aria-label="Language"> | |
| <button type="button" data-lang="en" aria-pressed="true">EN</button> | |
| <button type="button" data-lang="fr" aria-pressed="false">FR</button> | |
| </div> | |
| <a href="https://github.com/sxc3030-eng/mat-nexus-showcase" data-i18n="github">GitHub repository</a> | |
| </div> | |
| </nav> | |
| <div class="hero-copy"> | |
| <p class="eyebrow">PUBLIC EVIDENCE DASHBOARD</p> | |
| <h1 data-i18n="heroTitle">A verified layer can help a small LLM — within a measurable domain.</h1> | |
| <p class="lede" data-i18n="heroLead">Paired comparisons on mechanically verifiable tasks. Diagnostics, pretests and invalidated campaigns are never pooled with primary evidence.</p> | |
| <div class="status-line" id="data-status" role="status">Static audited snapshot loaded · 4 primary comparisons · data 2026-08-02</div> | |
| </div> | |
| </header> | |
| <main id="main"> | |
| <section aria-labelledby="headline-title"> | |
| <div class="section-heading"> | |
| <div><p class="eyebrow" data-i18n="primaryEyebrow">PRIMARY EVIDENCE</p><h2 id="headline-title" data-i18n="primaryTitle">LLM alone vs LLM + Nexus Safe</h2></div> | |
| <p data-i18n="primaryNote">Accuracy is directly labelled. No hover is required.</p> | |
| </div> | |
| <div class="kpi-grid" id="kpis" aria-label="Indicateurs principaux"> | |
| <div class="kpi"><strong>47</strong><span>audited questions / questions auditées</span></div> | |
| <div class="kpi"><strong>2/4</strong><span>comparisons classified as GAIN</span></div> | |
| <div class="kpi"><strong>+68.1 pp</strong><span>descriptive aggregate gap</span></div> | |
| </div> | |
| <div class="chart-panel"> | |
| <div class="legend" aria-label="Legend"><span><i class="swatch raw"></i><span data-i18n="rawLabel">LLM alone</span></span><span><i class="swatch nexus"></i><span data-i18n="nexusLabel">LLM + Nexus Safe</span></span></div> | |
| <div id="paired-chart" class="paired-chart" aria-live="polite"></div> | |
| </div> | |
| </section> | |
| <section aria-labelledby="decision-title"> | |
| <div class="section-heading"> | |
| <div><p class="eyebrow" data-i18n="decisionEyebrow">PAIRED DECISION</p><h2 id="decision-title" data-i18n="decisionTitle">Gain, loss or inconclusive result</h2></div> | |
| <p data-i18n="decisionNote">A higher score is not enough: the decision accounts for discordant pairs and sample size.</p> | |
| </div> | |
| <div class="decision-grid" id="decision-grid"> | |
| <article class="decision-card"><span class="badge GAIN">GAIN</span><h3>Granite 3.3 2B Instruct</h3><strong>15 wins · 0 losses</strong><p>Two-sided exact test: p=0.00006103515625. Status: AUDITED.</p></article> | |
| <article class="decision-card"><span class="badge GAIN">GAIN</span><h3>Gemma 3 12B IT QAT</h3><strong>12 wins · 0 losses</strong><p>Two-sided exact test: p=0.00048828125. Status: AUDITED.</p></article> | |
| <article class="decision-card"><span class="badge INCONCLUSIVE">INCONCLUSIVE</span><h3>Llama 3.1 8B</h3><strong>5 wins · 0 losses</strong><p>Two-sided exact test: p=0.0625. Status: AUDITED.</p></article> | |
| <article class="decision-card"><span class="badge INCONCLUSIVE">INCONCLUSIVE</span><h3>Granite 3.3 2B Instruct · code</h3><strong>0 wins · 0 losses</strong><p>Two-sided exact test: p=1. Status: AUDITED_SMALL_SAMPLE.</p></article> | |
| </div> | |
| </section> | |
| <section aria-labelledby="table-title"> | |
| <div class="section-heading"> | |
| <div><p class="eyebrow" data-i18n="portableEyebrow">PORTABLE DATA</p><h2 id="table-title" data-i18n="tableTitle">Complete primary comparison table</h2></div> | |
| <p><a href="data/public-benchmark-catalog.json">JSON</a> · <a href="data/primary-comparisons.csv">CSV</a></p> | |
| </div> | |
| <div class="table-wrap"> | |
| <table> | |
| <caption data-i18n="caption">Audited results published without questions, answers or targets.</caption> | |
| <thead><tr><th data-i18n="thModel">Model</th><th data-i18n="thDomain">Domain</th><th>n</th><th data-i18n="thAlone">Alone</th><th>Nexus</th><th>Δ</th><th data-i18n="thDecision">Decision</th><th data-i18n="thP">exact p</th></tr></thead> | |
| <tbody id="results-body"> | |
| <tr><td>Granite 3.3 2B Instruct</td><td>mixed verifiable</td><td>18</td><td>3/18 (16.7 %)</td><td>18/18 (100 %)</td><td>+83.3 pp</td><td><span class="badge GAIN">GAIN</span></td><td>0.00006103515625</td></tr> | |
| <tr><td>Gemma 3 12B IT QAT</td><td>mixed verifiable</td><td>18</td><td>3/18 (16.7 %)</td><td>15/18 (83.3 %)</td><td>+66.7 pp</td><td><span class="badge GAIN">GAIN</span></td><td>0.00048828125</td></tr> | |
| <tr><td>Llama 3.1 8B</td><td>mixed verifiable</td><td>6</td><td>1/6 (16.7 %)</td><td>6/6 (100 %)</td><td>+83.3 pp</td><td><span class="badge INCONCLUSIVE">INCONCLUSIVE</span></td><td>0.0625</td></tr> | |
| <tr><td>Granite 3.3 2B Instruct</td><td>code</td><td>5</td><td>1/5 (20 %)</td><td>1/5 (20 %)</td><td>+0.0 pp</td><td><span class="badge INCONCLUSIVE">INCONCLUSIVE</span></td><td>1</td></tr> | |
| </tbody> | |
| </table> | |
| </div> | |
| </section> | |
| <section aria-labelledby="audit-title"> | |
| <div class="section-heading"> | |
| <div><p class="eyebrow" data-i18n="auditEyebrow">AUDIT LEDGER</p><h2 id="audit-title" data-i18n="auditTitle">Informative evidence that is not a product claim</h2></div> | |
| <p data-i18n="auditNote">These campaigns remain visible, but are excluded from primary averages.</p> | |
| </div> | |
| <div class="audit-grid" id="audit-grid"></div> | |
| <h3 data-i18n="excludedTitle">Excluded or running campaigns</h3> | |
| <div class="excluded-list" id="excluded-list"></div> | |
| </section> | |
| <section class="method" aria-labelledby="method-title"> | |
| <p class="eyebrow" data-i18n="readingEyebrow">HOW TO READ THIS</p> | |
| <h2 id="method-title" data-i18n="methodTitle">What these numbers do — and do not — show</h2> | |
| <div class="method-grid"> | |
| <p data-i18n-html="shows"><strong>They show</strong> a reproducible gain on some verifiable problems when Nexus has an appropriate executor or verifier.</p> | |
| <p data-i18n-html="notShows"><strong>They do not show</strong> universal improvement in reasoning, creativity, code or open-domain knowledge.</p> | |
| <p data-i18n-html="publicationRule"><strong>Publication rule:</strong> previously observed answers are never reused to score a new campaign; sealed sets are not used for training.</p> | |
| </div> | |
| </section> | |
| </main> | |
| <footer><p data-i18n="footer">MAT Nexus public portfolio edition · aggregate data, explicit public boundaries.</p></footer> | |
| <noscript><p class="noscript">JavaScript is required to generate charts. Data remains available as JSON and CSV.</p></noscript> | |
| <script id="embedded-benchmark-catalog" type="application/json"> | |
| { | |
| "schema_version": "mat-nexus-public-benchmark-catalog-v1", | |
| "generated_on": "2026-08-02", | |
| "primary_comparisons": [ | |
| {"id":"short-18-granite","model":"Granite 3.3 2B Instruct","domain":"mixed_verifiable","questions":18,"raw_correct":3,"nexus_correct":18,"wins":15,"losses":0,"classification":"GAIN","two_sided_exact_p_value":0.00006103515625,"evidence_status":"AUDITED"}, | |
| {"id":"short-18-gemma","model":"Gemma 3 12B IT QAT","domain":"mixed_verifiable","questions":18,"raw_correct":3,"nexus_correct":15,"wins":12,"losses":0,"classification":"GAIN","two_sided_exact_p_value":0.00048828125,"evidence_status":"AUDITED"}, | |
| {"id":"micro-6-llama","model":"Llama 3.1 8B","domain":"mixed_verifiable","questions":6,"raw_correct":1,"nexus_correct":6,"wins":5,"losses":0,"classification":"INCONCLUSIVE","two_sided_exact_p_value":0.0625,"evidence_status":"AUDITED"}, | |
| {"id":"code-5-granite","model":"Granite 3.3 2B Instruct","domain":"code","questions":5,"raw_correct":1,"nexus_correct":1,"wins":0,"losses":0,"classification":"INCONCLUSIVE","two_sided_exact_p_value":1.0,"evidence_status":"AUDITED_SMALL_SAMPLE"} | |
| ], | |
| "diagnostic_campaigns": [ | |
| {"id":"nexus-four-arm-1000q-v1","questions":1000,"kind":"component_ablation","status":"HISTORICAL_DIAGNOSTIC","metrics":{"raw_accuracy":0.0,"prepared_accuracy":0.019,"experts_accuracy":0.812,"experts_memory_accuracy":0.905},"caveat":"Component ablation; not pooled with the current paired safe-layer benchmark."}, | |
| {"id":"nexus-adapter-ab-1000q-v1-safe","questions":1000,"kind":"adapter_nexus_ablation","status":"HISTORICAL_DIAGNOSTIC","metrics":{"base_raw_accuracy":0.0,"adapted_raw_accuracy":0.0,"base_nexus_accuracy":0.125,"adapted_nexus_accuracy":0.186},"caveat":"Useful for architecture diagnosis; raw-arm behavior makes it unsuitable as the headline gain claim."}, | |
| {"id":"full-circuit-all-functional-ab-v2-safe","questions":10,"kind":"controlled_preview","status":"EXPLORATORY","metrics":{"llm_alone_accuracy":0.0,"experts_raw_plus_llm_accuracy":0.0,"experts_prepared_plus_llm_accuracy":0.1},"caveat":"Controlled preview, explicitly not an official benchmark."} | |
| ], | |
| "excluded_campaigns": [ | |
| {"id":"nexus-neutral-10x20-v1","status":"INVALIDATED","reason":"Uniform 96-token ceiling truncated model outputs; 36 Granite cases were consumed and excluded from scoring."}, | |
| {"id":"nexus-neutral-10x20-v2","status":"RUNNING","reason":"Larger neutral paired campaign remains unpublished until completion and audit."} | |
| ] | |
| } | |
| </script> | |
| <script src="dashboard.js?v=5" defer></script> | |
| </body> | |
| </html> | |