joelniklaus HF Staff Cursor commited on
Commit
e75c23e
·
1 Parent(s): bd1aa3f

add COLM 2026 spotlight and Deep Learning with Yacine decks

Browse files
.gitattributes CHANGED
@@ -8,6 +8,7 @@
8
  *.csv filter=lfs diff=lfs merge=lfs -text
9
  *.json filter=lfs diff=lfs merge=lfs -text
10
  *.pdf filter=lfs diff=lfs merge=lfs -text
 
11
  *.wav filter=lfs diff=lfs merge=lfs -text
12
  *.mp3 filter=lfs diff=lfs merge=lfs -text
13
  # the package and package lock should not be tracked
 
8
  *.csv filter=lfs diff=lfs merge=lfs -text
9
  *.json filter=lfs diff=lfs merge=lfs -text
10
  *.pdf filter=lfs diff=lfs merge=lfs -text
11
+ *.pptx filter=lfs diff=lfs merge=lfs -text
12
  *.wav filter=lfs diff=lfs merge=lfs -text
13
  *.mp3 filter=lfs diff=lfs merge=lfs -text
14
  # the package and package lock should not be tracked
presentations/2026-06-18-deep-learning-with-yacine/finephrase-deep-learning-with-yacine.pptx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ee26336d4d63327c4f85147f1992c8ae190e7967230dd81e465447287ada3450
3
+ size 7545227
presentations/2026-10-colm-spotlight/assets/bookshelf.png ADDED

Git LFS Details

  • SHA256: 0d862e0aa54ba93913cd1a6b99ca0bff503ba8a803a707cb83a476ad35ca0489
  • Pointer size: 132 Bytes
  • Size of remote file: 1.98 MB
presentations/2026-10-colm-spotlight/assets/dataset-viewer.png ADDED

Git LFS Details

  • SHA256: 70b4f941bb30778e6de57b54838448b96865142d3b88c4e24843b311b267eb2c
  • Pointer size: 131 Bytes
  • Size of remote file: 404 kB
presentations/2026-10-colm-spotlight/assets/overview.jpg ADDED

Git LFS Details

  • SHA256: fc6eb8fdddf108454a3351e927dcc42536c241e92ae946efa7b2e5c5d629959b
  • Pointer size: 131 Bytes
  • Size of remote file: 172 kB
presentations/2026-10-colm-spotlight/assets/qr-code.svg ADDED
presentations/2026-10-colm-spotlight/assets/qr-dataset.svg ADDED
presentations/2026-10-colm-spotlight/assets/qr-paper.svg ADDED
presentations/2026-10-colm-spotlight/build.mjs ADDED
@@ -0,0 +1,596 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Build the COLM 2026 spotlight talk (12 min) for the FinePhrase paper.
4
+ *
5
+ * Usage (from this directory):
6
+ * npm install && node build.mjs
7
+ *
8
+ * All scores are macro averages (x100) over 12 benchmarks from the paper
9
+ * (paper/colm2026_conference.tex and paper/data/*.dat). Visual style follows
10
+ * the Deep Learning with Yacine deck: Calibri, slate/amber palette, dark title slides.
11
+ */
12
+ import { readFile } from "node:fs/promises";
13
+ import { dirname, resolve } from "node:path";
14
+ import { fileURLToPath } from "node:url";
15
+ import pptxgen from "pptxgenjs";
16
+ import sharp from "sharp";
17
+
18
+ const here = dirname(fileURLToPath(import.meta.url));
19
+ const asset = (name) => resolve(here, "assets", name);
20
+ const OUT = resolve(here, "finephrase-colm-2026-spotlight.pptx");
21
+
22
+ const FONT = "Calibri";
23
+ const C = {
24
+ navy: "0F172A",
25
+ ink: "1E293B",
26
+ muted: "64748B",
27
+ faint: "94A3B8",
28
+ line: "E2E8F0",
29
+ light: "F8FAFC",
30
+ cool: "F1F5F9",
31
+ amber: "F59E0B",
32
+ amberDark: "B45309",
33
+ amberSoft: "FFF7D6",
34
+ yellow: "FFD21E",
35
+ gray: "CBD5E1",
36
+ ref: "475569",
37
+ red: "F87171",
38
+ white: "FFFFFF",
39
+ };
40
+ const DCLM = 13.77;
41
+
42
+ const pres = new pptxgen();
43
+ pres.layout = "LAYOUT_16x9"; // 10in x 5.625in
44
+ pres.author = "Joel Niklaus";
45
+ pres.company = "Hugging Face";
46
+ pres.title = "How Can We Synthesize High-Quality Pretraining Data? (COLM 2026)";
47
+
48
+ // ---------------------------------------------------------------------------
49
+ // Helpers
50
+ // ---------------------------------------------------------------------------
51
+
52
+ /** Rasterize an SVG string or buffer to a pptxgenjs data URI. */
53
+ async function pngData(svg, { flatten = false } = {}) {
54
+ let img = sharp(Buffer.from(svg), { density: 300 });
55
+ if (flatten) img = img.flatten({ background: "#ffffff" });
56
+ return `image/png;base64,${(await img.png().toBuffer()).toString("base64")}`;
57
+ }
58
+
59
+ /** 1,000-cell waffle; the first `highlight` cells (column-major, so they form a block) are red. */
60
+ function waffleSvg(highlight, { cols = 50, rows = 20, cell = 10, gap = 3 } = {}) {
61
+ const step = cell + gap;
62
+ let rects = "";
63
+ for (let i = 0; i < cols * rows; i++) {
64
+ const [col, row] = [Math.floor(i / rows), i % rows];
65
+ const fill = i < highlight ? C.red : C.gray;
66
+ rects += `<rect x="${col * step}" y="${row * step}" width="${cell}" height="${cell}" rx="2" fill="#${fill}"/>`;
67
+ }
68
+ return `<svg xmlns="http://www.w3.org/2000/svg" width="${cols * step - gap}" height="${rows * step - gap}">${rects}</svg>`;
69
+ }
70
+
71
+ function text(slide, str, opts) {
72
+ slide.addText(str, { fontFace: FONT, margin: 0, color: C.ink, valign: "top", ...opts });
73
+ }
74
+
75
+ /** Amber kicker + bold title, the header used on every content slide. */
76
+ function header(slide, kicker, title, { w = 9, h = 0.8 } = {}) {
77
+ text(slide, kicker, { x: 0.5, y: 0.3, w, h: 0.25, fontSize: 11, bold: true, charSpacing: 1.5, color: C.amber });
78
+ text(slide, title, { x: 0.5, y: 0.58, w, h, fontSize: 26, bold: true, color: C.navy });
79
+ }
80
+
81
+ function footer(slide, n, total) {
82
+ text(slide, "FinePhrase · COLM 2026", { x: 0.5, y: 5.25, w: 5, h: 0.22, fontSize: 10, color: C.muted });
83
+ text(slide, `${n} / ${total}`, { x: 8.5, y: 5.25, w: 1, h: 0.22, fontSize: 10, color: C.muted, align: "right" });
84
+ }
85
+
86
+ /** Small colored square + label rows, used as manual chart legends. */
87
+ function legend(slide, items, { x, y, size = 12 }) {
88
+ items.forEach(([color, label], i) => {
89
+ slide.addShape(pres.shapes.RECTANGLE, { x, y: y + i * 0.34 + 0.04, w: 0.18, h: 0.18, fill: { color }, line: { color } });
90
+ text(slide, label, { x: x + 0.3, y: y + i * 0.34, w: 2.9, h: 0.26, fontSize: size, color: C.ink, valign: "middle" });
91
+ });
92
+ }
93
+
94
+ const axisStyle = () => ({
95
+ catAxisLabelFontFace: FONT,
96
+ catAxisLabelFontSize: 13,
97
+ catAxisLabelColor: C.ink,
98
+ catAxisLineShow: false,
99
+ catGridLine: { style: "none" },
100
+ valGridLine: { style: "none" },
101
+ valAxisHidden: true,
102
+ dataLabelFontFace: FONT,
103
+ dataLabelFontSize: 13,
104
+ dataLabelFontBold: true,
105
+ dataLabelColor: C.ink,
106
+ dataLabelFormatCode: "0.0",
107
+ });
108
+
109
+ /** Horizontal ranking bars, one color per bar, first row on top. */
110
+ function rankBars(slide, rows, box, { max, fontSize = 13 }) {
111
+ slide.addChart(
112
+ pres.charts.BAR,
113
+ [{ name: "Score", labels: rows.map((r) => r.label), values: rows.map((r) => r.value) }],
114
+ {
115
+ ...box,
116
+ ...axisStyle(),
117
+ barDir: "bar",
118
+ catAxisOrientation: "maxMin",
119
+ catAxisLabelFontSize: fontSize,
120
+ chartColors: rows.map((r) => r.color),
121
+ valAxisMinVal: 0,
122
+ valAxisMaxVal: max,
123
+ barGapWidthPct: 35,
124
+ showValue: true,
125
+ dataLabelPosition: "outEnd",
126
+ dataLabelFontSize: fontSize,
127
+ showLegend: false,
128
+ },
129
+ );
130
+ }
131
+
132
+ /** Clustered columns: gray baseline vs amber treatment. */
133
+ function pairedColumns(slide, labels, [baseName, baseValues], [ourName, ourValues], box, extra = {}) {
134
+ slide.addChart(
135
+ pres.charts.BAR,
136
+ [
137
+ { name: baseName, labels, values: baseValues },
138
+ { name: ourName, labels, values: ourValues },
139
+ ],
140
+ {
141
+ ...box,
142
+ ...axisStyle(),
143
+ barDir: "col",
144
+ barGrouping: "clustered",
145
+ chartColors: [C.gray, C.amber],
146
+ barGapWidthPct: 60,
147
+ showValue: true,
148
+ dataLabelPosition: "outEnd",
149
+ catAxisLabelFontSize: 15,
150
+ valAxisMinVal: 0,
151
+ valAxisMaxVal: 17,
152
+ showLegend: true,
153
+ legendPos: "b",
154
+ legendFontFace: FONT,
155
+ legendFontSize: 13,
156
+ legendColor: C.ink,
157
+ ...extra,
158
+ },
159
+ );
160
+ }
161
+
162
+ const ours = (label, value) => ({ label, value, color: C.amber });
163
+ const prior = (label, value) => ({ label, value, color: C.gray });
164
+ const dclm = { label: "DCLM (web)", value: DCLM, color: C.ref };
165
+
166
+ // ---------------------------------------------------------------------------
167
+ // Slides. Each function draws one slide; the loop at the bottom adds footers.
168
+ // ---------------------------------------------------------------------------
169
+
170
+ function titleSlide(s) {
171
+ s.background = { color: C.navy };
172
+ text(s, "FINEPHRASE", { x: 0.5, y: 1.2, w: 4.4, h: 0.3, fontSize: 14, bold: true, charSpacing: 2, color: C.yellow });
173
+ text(s, "How Can We Synthesize High-Quality Pretraining Data?", { x: 0.5, y: 1.55, w: 4.4, h: 1.8, fontSize: 30, bold: true, color: C.white });
174
+ text(s, "Joel Niklaus · Hugging Face", { x: 0.5, y: 3.45, w: 4.4, h: 0.32, fontSize: 15, color: C.white });
175
+ text(s, "COLM 2026 · San Francisco", { x: 0.5, y: 3.8, w: 4.4, h: 0.3, fontSize: 12, color: C.yellow });
176
+ s.addImage({ path: asset("bookshelf.png"), x: 5.05, y: 1.3, w: 4.45, h: 2.23 });
177
+ text(
178
+ s,
179
+ [
180
+ { text: "With Atsuki Yamaguchi, Michal Štefánik, Guilherme Penedo, Hynek Kydlíček, Elie Bakouch, Lewis Tunstall, Edward Beeching, Thibaud Frere, Colin Raffel, Leandro von Werra, Thomas Wolf", options: { breakLine: true } },
181
+ { text: "Hugging Face · University of Sheffield · National Institute of Informatics, Japan" },
182
+ ],
183
+ { x: 0.5, y: 4.6, w: 9, h: 0.5, fontSize: 8, color: C.faint, paraSpaceAfter: 2 },
184
+ );
185
+ s.addNotes(
186
+ "~20s. Hi, I'm Joel from Hugging Face. This is joint work with colleagues from Hugging Face, the University of Sheffield and NII. " +
187
+ "The question is simple: how do you synthesize high-quality pretraining data? We answer it with a controlled study over prompts, generator models and data, " +
188
+ "and turn the findings into FinePhrase, a 486B-token open dataset.",
189
+ );
190
+ }
191
+
192
+ function scaleOfSyntheticData(s) {
193
+ text(s, "2T", { x: 0.5, y: 0.7, w: 9, h: 2.5, fontSize: 160, bold: true, color: C.navy, align: "center", valign: "middle" });
194
+ text(s, "tokens of web text rephrased for Nemotron-CC", { x: 0.5, y: 3.3, w: 9, h: 0.5, fontSize: 22, bold: true, color: C.navy, align: "center" });
195
+ text(s, "Phi-4 · Qwen3 · Nemotron 3 · Arcee Trinity all pretrain on synthetic data", { x: 0.5, y: 3.85, w: 9, h: 0.4, fontSize: 14, color: C.muted, align: "center" });
196
+ s.addNotes(
197
+ "~25s. Synthetic data is now a standard pretraining ingredient. NVIDIA rephrased two trillion tokens of web text for Nemotron-CC, " +
198
+ "and recent models like Phi-4, Qwen3, Nemotron 3 and Arcee Trinity pretrain on hundreds of billions of synthetic tokens.",
199
+ );
200
+ }
201
+
202
+ function theGap(s) {
203
+ header(s, "THE GAP", "Every recipe was tested against its own baseline");
204
+ const cards = [
205
+ ["WRAP", "style rewrites"],
206
+ ["Nemotron-CC", "QA pairs & knowledge lists"],
207
+ ["BeyondWeb", "continue & summarize"],
208
+ ["REWIRE", "guided rewrites"],
209
+ ];
210
+ cards.forEach(([name, approach], i) => {
211
+ const x = 0.5 + i * 2.333;
212
+ s.addShape(pres.shapes.ROUNDED_RECTANGLE, { x, y: 1.55, w: 2.0, h: 2.45, fill: { color: C.light }, line: { color: C.line }, rectRadius: 0.08 });
213
+ text(s, name, { x: x + 0.2, y: 1.72, w: 1.6, h: 0.4, fontSize: 18, bold: true, color: C.navy });
214
+ text(s, approach, { x: x + 0.2, y: 2.12, w: 1.6, h: 0.3, fontSize: 12, color: C.muted });
215
+ // Mini "method beats its own baseline" bars: every paper reports a win, none against the same reference.
216
+ s.addShape(pres.shapes.RECTANGLE, { x: x + 0.45, y: 3.05, w: 0.4, h: 0.6, fill: { color: C.gray }, line: { color: C.gray } });
217
+ s.addShape(pres.shapes.RECTANGLE, { x: x + 1.15, y: 2.65, w: 0.4, h: 1.0, fill: { color: C.amber }, line: { color: C.amber } });
218
+ text(s, "own baseline", { x: x + 0.15, y: 3.7, w: 1.0, h: 0.22, fontSize: 9, color: C.muted, align: "center" });
219
+ text(s, "method", { x: x + 0.85, y: 3.7, w: 1.0, h: 0.22, fontSize: 9, color: C.muted, align: "center" });
220
+ });
221
+ text(s, "Which prompt? Which generator? Which data?", { x: 0.5, y: 4.35, w: 9, h: 0.45, fontSize: 20, bold: true, italic: true, color: C.amber, align: "center" });
222
+ s.addNotes(
223
+ "~35s. But the design space is huge and every method was evaluated in isolation. WRAP does style rewrites, Nemotron-CC extracts QA pairs and knowledge lists, " +
224
+ "BeyondWeb continues and summarizes, REWIRE does guided rewrites of low-quality pages. Each paper reports a gain against its own baseline, with its own generator and its own data. " +
225
+ "So if you want to build synthetic data today, you don't know which prompt, which generator, or which data to use.",
226
+ );
227
+ }
228
+
229
+ function studyDesign(s) {
230
+ header(s, "OUR STUDY", "One controlled setup, three axes");
231
+ s.addImage({ path: asset("overview.jpg"), x: 1.5, y: 1.3, w: 7.0, h: 3.42 });
232
+ text(s, "Rephrase ~10.5B tokens · pretrain a 1.2B model on 21B tokens · evaluate on 12 benchmarks", {
233
+ x: 0.5, y: 4.82, w: 9, h: 0.3, fontSize: 13, color: C.muted, align: "center",
234
+ });
235
+ s.addNotes(
236
+ "~45s. We fix one pipeline and change one component at a time. Three axes: the rephrasing prompt P, the generator model G, and the data, " +
237
+ "meaning the source we rephrase and the original web data we mix in. Each configuration rephrases about 10.5B tokens, we pretrain a 1.2B model from scratch on 21B tokens, " +
238
+ "and evaluate on 12 benchmarks with 3-shot cloze prompts. I'll report the macro average. The reference throughout is DCLM, the best curated web dataset we tested, at 13.8.",
239
+ );
240
+ }
241
+
242
+ function studyScale(s) {
243
+ const stats = [
244
+ ["90", "rephrasing configurations"],
245
+ ["333", "train-and-evaluate runs"],
246
+ ["1T+", "synthetic tokens generated"],
247
+ ["12.7", "GPU-years of generation"],
248
+ ];
249
+ stats.forEach(([n, label], i) => {
250
+ const x = 0.5 + i * 2.25;
251
+ text(s, n, { x, y: 1.55, w: 2.25, h: 1.3, fontSize: 66, bold: true, color: C.navy, align: "center", valign: "bottom" });
252
+ text(s, label, { x: x + 0.15, y: 3.0, w: 1.95, h: 0.7, fontSize: 15, color: C.muted, align: "center" });
253
+ });
254
+ s.addNotes("~20s. To get robust answers we ran this at scale: 90 rephrasing configurations, 333 train-and-evaluate runs, over one trillion synthetic tokens, 12.7 GPU-years of generation.");
255
+ }
256
+
257
+ function formats(s) {
258
+ header(s, "AXIS 1 · PROMPT", "One web page, four pedagogical formats", { w: 3.7, h: 1.3 });
259
+ const chips = [
260
+ ["FAQ", "4EA8DE", "questions from basic to advanced"],
261
+ ["MATH", "A5A0F0", "word problem + worked solution"],
262
+ ["TABLE", "B07CC6", "structured table + Q&A pair"],
263
+ ["TUTORIAL", "E08DB8", "numbered step-by-step guide"],
264
+ ];
265
+ chips.forEach(([name, color, desc], i) => {
266
+ const y = 2.1 + i * 0.62;
267
+ s.addShape(pres.shapes.ROUNDED_RECTANGLE, { x: 0.5, y, w: 1.15, h: 0.4, fill: { color }, line: { color }, rectRadius: 0.08 });
268
+ text(s, name, { x: 0.5, y, w: 1.15, h: 0.4, fontSize: 12, bold: true, color: C.white, align: "center", valign: "middle", charSpacing: 1 });
269
+ text(s, desc, { x: 1.8, y, w: 2.4, h: 0.4, fontSize: 12, color: C.muted, valign: "middle" });
270
+ });
271
+ s.addImage({ path: asset("dataset-viewer.png"), x: 4.55, y: 0.35, w: 4.95, h: 4.365 });
272
+ s.addNotes(
273
+ "~40s. Axis one: the prompt. Besides eight prompts from prior work, we designed four pedagogical formats that restructure the page instead of paraphrasing it. " +
274
+ "Here is one FineWeb-Edu page about grape species, rewritten as a FAQ, a math word problem, a table with a question-answer pair, and a step-by-step tutorial.",
275
+ );
276
+ }
277
+
278
+ function promptResults(s) {
279
+ header(s, "AXIS 1 · PROMPT", "Structured formats beat paraphrasing", { w: 3.2, h: 1.3 });
280
+ legend(s, [[C.amber, "Our pedagogical formats"], [C.ref, "DCLM (curated web)"], [C.gray, "Prompts from prior work"]], { x: 0.5, y: 2.2 });
281
+ text(s, "Gemma 3 1B generator · macro score over 12 benchmarks", { x: 0.5, y: 3.45, w: 3.0, h: 0.5, fontSize: 11, color: C.muted });
282
+ rankBars(
283
+ s,
284
+ [
285
+ ours("Math", 15.31),
286
+ ours("Table", 14.83),
287
+ prior("Diverse QA Pairs", 14.58),
288
+ ours("FAQ", 14.45),
289
+ ours("Tutorial", 14.3),
290
+ dclm,
291
+ prior("Continue", 13.73),
292
+ prior("Guided Rewrite", 13.72),
293
+ prior("Distill", 13.17),
294
+ prior("Wikipedia", 13.14),
295
+ prior("Knowledge List", 13.11),
296
+ prior("Summarize", 13.01),
297
+ prior("Extract Knowledge", 11.81),
298
+ ],
299
+ { x: 3.75, y: 0.3, w: 5.9, h: 4.85 },
300
+ { max: 17.5, fontSize: 12 },
301
+ );
302
+ s.addNotes(
303
+ "~50s. All four structured formats beat DCLM, and Math and Table beat every prompt from prior work. Math is +1.5 over DCLM. " +
304
+ "The only prior prompt that competes is Diverse QA Pairs from Nemotron-CC, which is also a structured question-answer format. " +
305
+ "Plain restatements like Distill, Wikipedia-style or Summarize land below DCLM. Restructure, don't paraphrase.",
306
+ );
307
+ }
308
+
309
+ function generatorSize(s) {
310
+ header(s, "AXIS 2 · GENERATOR SIZE", "Past 1B, bigger generators don't help");
311
+ const labels = ["270M", "1B", "4B", "12B", "27B"];
312
+ s.addChart(
313
+ [
314
+ {
315
+ type: pres.charts.LINE,
316
+ data: [{ name: "Gemma 3 generator (Math prompt)", labels, values: [13.8, 15.31, 15.06, 14.68, 14.76] }],
317
+ options: { chartColors: [C.amber], lineSize: 3, lineDataSymbol: "circle", lineDataSymbolSize: 11, showValue: true, dataLabelPosition: "t" },
318
+ },
319
+ {
320
+ type: pres.charts.LINE,
321
+ data: [{ name: "DCLM (curated web)", labels, values: labels.map(() => DCLM) }],
322
+ // Two entries: pptxgenjs picks marker colors by overall series index and falls back to a random color.
323
+ options: { chartColors: [C.ref, C.ref], lineSize: 2, lineDash: "dash", lineDataSymbol: "none", showValue: false },
324
+ },
325
+ ],
326
+ {
327
+ x: 0.8, y: 1.35, w: 8.4, h: 3.8,
328
+ ...axisStyle(),
329
+ catAxisLabelFontSize: 16,
330
+ catAxisLineShow: true,
331
+ catAxisLineColor: C.line,
332
+ dataLabelFontSize: 15,
333
+ valAxisMinVal: 12.5,
334
+ valAxisMaxVal: 16,
335
+ showLegend: true,
336
+ legendPos: "b",
337
+ legendFontFace: FONT,
338
+ legendFontSize: 13,
339
+ legendColor: C.ink,
340
+ },
341
+ );
342
+ s.addNotes(
343
+ "~45s. Axis two: the generator. We rephrase with Gemma 3 from 270M to 27B using the Math prompt. 270M is too small and barely matches DCLM. " +
344
+ "From 1B up the curve is flat: 1B is the best at 15.3, 27B lands at 14.8. The one exception we found is REWIRE's complex guided rewrite prompt, where 4B helps. " +
345
+ "Meanwhile 12B and 27B cost 5 to 10 times more GPU time.",
346
+ );
347
+ }
348
+
349
+ function generatorFamily(s) {
350
+ header(s, "AXIS 2 · GENERATOR FAMILY", "SmolLM2 is the best rephraser", { w: 3.2, h: 1.3 });
351
+ text(s, "~1B instruct models · averaged over the four formats", { x: 0.5, y: 2.1, w: 3.0, h: 0.5, fontSize: 11, color: C.muted });
352
+ rankBars(
353
+ s,
354
+ [
355
+ ours("SmolLM2 1.7B", 16.55),
356
+ prior("Falcon 3 1B", 15.54),
357
+ prior("Granite 3.1 1B", 14.87),
358
+ prior("Llama 3.2 1B", 14.79),
359
+ prior("Gemma 3 1B", 14.72),
360
+ prior("Qwen3 1.7B", 14.49),
361
+ dclm,
362
+ ],
363
+ { x: 3.75, y: 0.5, w: 5.9, h: 4.5 },
364
+ { max: 18.5, fontSize: 14 },
365
+ );
366
+ s.addNotes(
367
+ "~35s. The family matters more than the size. At around 1B, averaged over the four formats, SmolLM2 1.7B beats every other family by 1 to 2 points. " +
368
+ "Most of that comes from reading comprehension, over 5 points on SQuAD. Our guess is SmolLM2's instruction data, which contains explicit rewrite tasks.",
369
+ );
370
+ }
371
+
372
+ async function diversity(s) {
373
+ header(s, "WHY SMOLLM2?", "Diversity beats polish");
374
+ const cols = { name: 0.5, fmt: 2.75, waffle: 4.1, score: 7.9 };
375
+ const head = { y: 1.38, h: 0.3, fontSize: 11, color: C.muted, align: "center" };
376
+ text(s, "follows format", { ...head, x: cols.fmt, w: 1.25 });
377
+ text(s, "1,000 math outputs · red = most common opening", { ...head, x: cols.waffle, w: 3.6 });
378
+ text(s, "macro score", { ...head, x: cols.score, w: 1.6 });
379
+ const rows = [
380
+ ["Qwen3 1.7B", "115 identical openings", "100%", 115, "14.4", C.ink],
381
+ ["SmolLM2 1.7B", "at most 3 identical openings", "68%", 3, "17.0", C.amber],
382
+ ];
383
+ for (const [i, [name, sub, fmt, same, score, scoreColor]] of rows.entries()) {
384
+ const y = 1.78 + i * 1.7;
385
+ text(s, name, { x: cols.name, y: y + 0.4, w: 2.2, h: 0.4, fontSize: 18, bold: true, color: C.navy });
386
+ text(s, sub, { x: cols.name, y: y + 0.8, w: 2.2, h: 0.3, fontSize: 12, color: same > 10 ? C.red : C.muted });
387
+ text(s, fmt, { x: cols.fmt, y, w: 1.25, h: 1.43, fontSize: 30, bold: true, color: C.navy, align: "center", valign: "middle" });
388
+ s.addImage({ data: await pngData(waffleSvg(same)), x: cols.waffle, y, w: 3.6, h: 1.43 });
389
+ text(s, score, { x: cols.score, y, w: 1.6, h: 1.43, fontSize: 40, bold: true, color: scoreColor, align: "center", valign: "middle" });
390
+ }
391
+ s.addNotes(
392
+ "~50s. Why does SmolLM2 win? We compared 1,000 Math outputs from SmolLM2 and Qwen3. Qwen3 is the perfect student: 100% of its outputs have a clean problem and solution section. " +
393
+ "SmolLM2 completes only 68%. But Qwen3 collapses into a template: 115 of the 1,000 outputs start with identical text, while SmolLM2's most common opening appears 3 times. " +
394
+ "Downstream, SmolLM2's messier data wins, 17.0 versus 14.4. For pretraining, output diversity matters more than format polish.",
395
+ );
396
+ }
397
+
398
+ function tradeOff(s) {
399
+ header(s, "THE TRADE-OFF", "Synthetic adds knowledge, web keeps commonsense");
400
+ const panels = [
401
+ { x: 0.5, w: 5.3, fill: C.amberSoft, title: "Knowledge & reading", titleColor: C.amberDark, stats: [["+18.7", "SQuAD v2"], ["+7.5", "TriviaQA"], ["+6.8", "ARC"]] },
402
+ { x: 6.05, w: 3.45, fill: C.cool, title: "Commonsense", titleColor: C.ref, stats: [["−1.6", "HellaSwag"], ["−1.1", "PIQA"]] },
403
+ ];
404
+ for (const p of panels) {
405
+ s.addShape(pres.shapes.ROUNDED_RECTANGLE, { x: p.x, y: 1.5, w: p.w, h: 3.1, fill: { color: p.fill }, line: { color: p.fill }, rectRadius: 0.1 });
406
+ text(s, p.title, { x: p.x + 0.3, y: 1.72, w: p.w - 0.6, h: 0.35, fontSize: 16, bold: true, color: p.titleColor });
407
+ const colW = (p.w - 0.6) / p.stats.length;
408
+ p.stats.forEach(([n, label], i) => {
409
+ const x = p.x + 0.3 + i * colW;
410
+ text(s, n, { x, y: 2.45, w: colW, h: 0.9, fontSize: 40, bold: true, color: C.navy, align: "center", valign: "middle" });
411
+ text(s, label, { x, y: 3.45, w: colW, h: 0.35, fontSize: 15, color: C.ink, align: "center" });
412
+ });
413
+ }
414
+ text(s, "Table format (SmolLM2 1.7B) minus DCLM, in benchmark points", { x: 0.5, y: 4.78, w: 9, h: 0.3, fontSize: 11, color: C.muted, align: "center" });
415
+ s.addNotes(
416
+ "~40s. Axis three is the data, and first a nuance: rephrased data is not better everywhere. Against DCLM, the Table format adds 18.7 points on SQuAD and around 7 on TriviaQA and ARC, " +
417
+ "but it loses on HellaSwag and PIQA, the commonsense tasks. Web text is full of everyday situations that restructuring strips out.",
418
+ );
419
+ }
420
+
421
+ function alwaysMix(s) {
422
+ header(s, "AXIS 3 · MIX-IN DATA", "Always mix in original web data");
423
+ pairedColumns(
424
+ s,
425
+ ["Math", "Table", "FAQ", "Tutorial"],
426
+ ["Synthetic only", [15.2, 13.74, 13.12, 12.32]],
427
+ ["50% synthetic + 50% web", [15.31, 14.83, 14.45, 14.3]],
428
+ { x: 0.8, y: 1.35, w: 8.4, h: 3.8 },
429
+ );
430
+ s.addNotes(
431
+ "~35s. So always mix. For every format, training on synthetic data alone loses to a 50/50 mix with original web text. Tutorial drops 2 points to 12.3, below DCLM at 13.8. " +
432
+ "The web half restores the commonsense and language understanding skills we just saw.",
433
+ );
434
+ }
435
+
436
+ function mixInOverSource(s) {
437
+ header(s, "AXIS 3 · SOURCE DATA", "A strong mix-in rescues weak sources");
438
+ pairedColumns(
439
+ s,
440
+ ["FineWeb-LQ", "Cosmopedia", "DCLM", "FineWeb-HQ"],
441
+ ["Mixed with the source itself", [9.63, 10.36, 13.69, 14.3]],
442
+ ["Mixed with FineWeb-HQ", [12.99, 13.88, 14.77, 14.3]],
443
+ { x: 0.8, y: 1.35, w: 8.4, h: 3.8 },
444
+ { showCatAxisTitle: true, catAxisTitle: "Source data that gets rephrased (Tutorial prompt)", catAxisTitleFontSize: 12, catAxisTitleColor: C.muted, catAxisTitleFontFace: FONT },
445
+ );
446
+ s.addNotes(
447
+ "~40s. Which data you mix in matters more than which data you rephrase. If the mix-in is the source itself, low-quality FineWeb only reaches 9.6. " +
448
+ "Rephrase the same low-quality source but mix in high-quality FineWeb, and you get 13.0. The spread across sources shrinks from 4.7 to 1.8 points. " +
449
+ "So you can up-cycle low-quality web text, which enlarges the pool of usable source data.",
450
+ );
451
+ }
452
+
453
+ function finephraseRecipe(s) {
454
+ s.background = { color: C.navy };
455
+ text(s, "PUTTING IT TOGETHER", { x: 0.5, y: 0.85, w: 9, h: 0.3, fontSize: 14, bold: true, charSpacing: 2, color: C.yellow });
456
+ text(s, "FinePhrase", { x: 0.5, y: 1.15, w: 9, h: 0.95, fontSize: 54, bold: true, color: C.white });
457
+ text(s, "486B tokens · 1.35B samples · open", { x: 0.5, y: 2.1, w: 9, h: 0.45, fontSize: 20, color: C.yellow });
458
+ const tiles = [
459
+ ["GENERATOR", "SmolLM2 1.7B"],
460
+ ["FORMATS", "Table · Math · FAQ · Tutorial"],
461
+ ["SOURCE", "FineWeb-Edu"],
462
+ ["MIX-IN", "FineWeb-HQ"],
463
+ ];
464
+ tiles.forEach(([label, value], i) => {
465
+ const x = 0.5 + i * 2.3;
466
+ s.addShape(pres.shapes.ROUNDED_RECTANGLE, { x, y: 3.0, w: 2.1, h: 1.45, fill: { color: C.ink }, line: { color: C.ink }, rectRadius: 0.08 });
467
+ text(s, label, { x: x + 0.2, y: 3.2, w: 1.7, h: 0.25, fontSize: 10, bold: true, charSpacing: 1.5, color: C.yellow });
468
+ text(s, value, { x: x + 0.2, y: 3.5, w: 1.7, h: 0.8, fontSize: 17, bold: true, color: C.white });
469
+ });
470
+ s.addNotes(
471
+ "~35s. Putting it together gives FinePhrase: SmolLM2 1.7B as the generator, the four pedagogical formats, FineWeb-Edu as the source, and FineWeb-HQ as the mix-in for training. " +
472
+ "That's 1.35 billion samples and 486 billion tokens, all open.",
473
+ );
474
+ }
475
+
476
+ function finephraseResults(s) {
477
+ header(s, "HEADLINE RESULT", "FinePhrase beats every synthetic baseline", { w: 3.2, h: 1.3 });
478
+ text(s, "+3.6", { x: 0.5, y: 2.15, w: 3.1, h: 1.0, fontSize: 60, bold: true, color: C.amber, valign: "middle" });
479
+ text(s, "points over the best synthetic baseline, Nemotron-HQ-Synth", { x: 0.5, y: 3.2, w: 3.0, h: 0.6, fontSize: 13, color: C.muted });
480
+ rankBars(
481
+ s,
482
+ [
483
+ ours("FinePhrase Table", 17.18),
484
+ ours("FinePhrase Math", 16.97),
485
+ ours("FinePhrase FAQ", 16.18),
486
+ ours("FinePhrase Tutorial", 15.88),
487
+ dclm,
488
+ prior("Nemotron-HQ-Synth", 13.54),
489
+ prior("REWIRE", 13.49),
490
+ prior("Cosmopedia", 10.33),
491
+ prior("SYNTH", 10.03),
492
+ ],
493
+ { x: 3.75, y: 0.4, w: 5.9, h: 4.7 },
494
+ { max: 19.5, fontSize: 13 },
495
+ );
496
+ s.addNotes(
497
+ "~45s. All four FinePhrase subsets beat every synthetic baseline. The Table subset reaches 17.2, that's +3.6 over Nemotron-HQ-Synth and +3.4 over DCLM. " +
498
+ "REWIRE, Cosmopedia and SYNTH are further behind. The per-benchmark profile is the one from before: FinePhrase dominates knowledge and reading comprehension, " +
499
+ "DCLM keeps a small edge on PIQA and HellaSwag, so FinePhrase is meant to be mixed with web data.",
500
+ );
501
+ }
502
+
503
+ function cost(s) {
504
+ header(s, "COST", "A 1.7B generator makes it cheap", { w: 3.2, h: 1.3 });
505
+ text(s, "30×", { x: 0.5, y: 2.15, w: 3.1, h: 1.1, fontSize: 72, bold: true, color: C.amber, valign: "middle" });
506
+ text(s, "more tokens per GPU-hour than REWIRE", { x: 0.5, y: 3.3, w: 3.0, h: 0.6, fontSize: 13, color: C.muted });
507
+ text(s, "Million tokens per H100 GPU-hour", { x: 3.85, y: 0.9, w: 5.6, h: 0.3, fontSize: 12, color: C.muted });
508
+ rankBars(
509
+ s,
510
+ [
511
+ ours("FinePhrase (SmolLM2 1.7B)", 33.1),
512
+ prior("SYNTH (fine-tuned)", 20),
513
+ prior("Cosmopedia (Mixtral 8x7B)", 2.5),
514
+ prior("REWIRE (Llama 3.3 70B)", 1.1),
515
+ ],
516
+ { x: 3.75, y: 1.2, w: 5.9, h: 3.4 },
517
+ { max: 38, fontSize: 14 },
518
+ );
519
+ s.addNotes(
520
+ "~30s. And it's cheap. With a 1.7B generator and speculative decoding we get about 9,200 tokens per second per H100. The full dataset took about 14,700 GPU-hours on 100 GPUs. " +
521
+ "That's 30 times more tokens per GPU-hour than REWIRE with Llama 3.3 70B and about 13 times more than Cosmopedia. SYNTH is also cheap, but its data trains much worse.",
522
+ );
523
+ }
524
+
525
+ function takeaways(s) {
526
+ header(s, "TAKEAWAYS", "What matters for synthetic pretraining data");
527
+ const items = [
528
+ ["The prompt matters most", "Structured formats beat paraphrasing"],
529
+ ["1B generators are enough", "Bigger models cost more and help little"],
530
+ ["Diversity beats polish", "Template collapse hurts pretraining"],
531
+ ["Always mix in web data", "The mix-in matters more than the source"],
532
+ ];
533
+ items.forEach(([head, sub], i) => {
534
+ const x = 0.5 + (i % 2) * 4.6;
535
+ const y = 1.5 + Math.floor(i / 2) * 1.72;
536
+ s.addShape(pres.shapes.ROUNDED_RECTANGLE, { x, y, w: 4.4, h: 1.52, fill: { color: C.light }, line: { color: C.line }, rectRadius: 0.08 });
537
+ s.addShape(pres.shapes.OVAL, { x: x + 0.3, y: y + 0.3, w: 0.52, h: 0.52, fill: { color: C.amber }, line: { color: C.amber } });
538
+ text(s, String(i + 1), { x: x + 0.3, y: y + 0.3, w: 0.52, h: 0.52, fontSize: 18, bold: true, color: C.navy, align: "center", valign: "middle" });
539
+ text(s, head, { x: x + 1.05, y: y + 0.3, w: 3.15, h: 0.45, fontSize: 18, bold: true, color: C.navy });
540
+ text(s, sub, { x: x + 1.05, y: y + 0.8, w: 3.15, h: 0.4, fontSize: 13, color: C.muted });
541
+ });
542
+ s.addNotes("~30s. Four takeaways. The prompt matters most. 1B generators are enough. Diversity beats polish. And always mix with strong web data, because the mix-in matters more than the source.");
543
+ }
544
+
545
+ async function thankYou(s) {
546
+ s.background = { color: C.navy };
547
+ text(s, "Thank you.", { x: 0.5, y: 0.8, w: 9, h: 1.0, fontSize: 54, bold: true, color: C.white });
548
+ text(s, "Questions?", { x: 0.5, y: 1.8, w: 9, h: 0.5, fontSize: 22, color: C.yellow });
549
+ text(s, "Come to our poster", { x: 0.5, y: 3.0, w: 4.3, h: 0.3, fontSize: 13, color: C.faint });
550
+ text(s, "F-4-126 · Franciscan · Wednesday afternoon", { x: 0.5, y: 3.3, w: 4.5, h: 0.4, fontSize: 16, bold: true, color: C.yellow });
551
+ text(s, "joel@niklaus.ai", { x: 0.5, y: 4.15, w: 4.3, h: 0.3, fontSize: 13, color: C.white });
552
+ const qrs = [
553
+ ["qr-paper.svg", "Paper", "arXiv 2604.13977"],
554
+ ["qr-dataset.svg", "Dataset", "HuggingFaceFW/finephrase"],
555
+ ["qr-code.svg", "Prompts + code", "huggingface/finephrase"],
556
+ ];
557
+ for (const [i, [file, label, sub]] of qrs.entries()) {
558
+ const x = 5.35 + i * 1.45;
559
+ s.addShape(pres.shapes.ROUNDED_RECTANGLE, { x, y: 2.6, w: 1.25, h: 1.25, fill: { color: C.white }, line: { color: C.white }, rectRadius: 0.08 });
560
+ s.addImage({ data: await pngData(await readFile(asset(file)), { flatten: true }), x: x + 0.08, y: 2.68, w: 1.09, h: 1.09 });
561
+ text(s, label, { x: x - 0.1, y: 3.95, w: 1.45, h: 0.28, fontSize: 12, bold: true, color: C.white, align: "center" });
562
+ text(s, sub, { x: x - 0.1, y: 4.23, w: 1.45, h: 0.25, fontSize: 8, color: C.faint, align: "center" });
563
+ }
564
+ s.addNotes("~20s. The dataset, all prompts and the generation code are open, the QR codes link to them. Come by the poster on Wednesday afternoon. Thank you!");
565
+ }
566
+
567
+ // [draw function, has footer]
568
+ const slides = [
569
+ [titleSlide, false],
570
+ [scaleOfSyntheticData, true],
571
+ [theGap, true],
572
+ [studyDesign, true],
573
+ [studyScale, true],
574
+ [formats, true],
575
+ [promptResults, true],
576
+ [generatorSize, true],
577
+ [generatorFamily, true],
578
+ [diversity, true],
579
+ [tradeOff, true],
580
+ [alwaysMix, true],
581
+ [mixInOverSource, true],
582
+ [finephraseRecipe, true],
583
+ [finephraseResults, true],
584
+ [cost, true],
585
+ [takeaways, true],
586
+ [thankYou, false],
587
+ ];
588
+
589
+ for (const [i, [draw, withFooter]] of slides.entries()) {
590
+ const slide = pres.addSlide();
591
+ await draw(slide);
592
+ if (withFooter) footer(slide, i + 1, slides.length);
593
+ }
594
+
595
+ await pres.writeFile({ fileName: OUT });
596
+ console.log(`Wrote ${OUT}`);
presentations/2026-10-colm-spotlight/finephrase-colm-2026-spotlight.pptx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e1d3881acbe863a91db19d4be3bc38ecfd3b0e1890f5c2ce0d76eae16aebb493
3
+ size 3736681
presentations/2026-10-colm-spotlight/package-lock.json ADDED
Binary file (27.3 kB). View file
 
presentations/2026-10-colm-spotlight/package.json ADDED
Binary file (211 Bytes). View file
 
presentations/README.md ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Presentations
2
+
3
+ Talks about FinePhrase, one folder per talk (`YYYY-MM[-DD]-<venue>`). The COLM 2026 poster lives in `../poster/`.
4
+
5
+ | Folder | Talk | Length | Slides |
6
+ | -------------------------------------- | ------------------------------------------ | ------- | ------ |
7
+ | `2026-02-26-se2026-industry-day` | SE 2026 Industry Day, Bern (reveal.js) | 30 min | 21 |
8
+ | `2026-06-18-deep-learning-with-yacine` | Deep Learning with Yacine (full deep dive) | ~50 min | 80 |
9
+ | `2026-10-colm-spotlight` | COLM 2026 spotlight oral, San Francisco | 12 min | 18 |
10
+
11
+ ## SE 2026 Industry Day
12
+
13
+ `index.html` is a [reveal.js](https://revealjs.com/) deck whose D3 charts (`charts/`) load `data/` at runtime, so serve the folder over HTTP to present it (`python3 -m http.server`). `build-standalone.mjs` inlines the CSS, images, charts and data into `standalone.html`, a single file you can open straight from disk (reveal.js and D3 still come from a CDN):
14
+
15
+ ```bash
16
+ cd presentations/2026-02-26-se2026-industry-day
17
+ node build-standalone.mjs
18
+ ```
19
+
20
+ ## COLM 2026 spotlight
21
+
22
+ `build.mjs` generates `finephrase-colm-2026-spotlight.pptx` with [pptxgenjs](https://gitbrent.github.io/PptxGenJS/). All charts are native PowerPoint charts with scores from the paper (`paper/colm2026_conference.tex`, `paper/data/*.dat`), and every slide has speaker notes with a target duration (about 10:40 in total). `assets/` holds the paper overview figure, the bookshelf and dataset-viewer images from the Yacine deck, and the QR codes from the poster.
23
+
24
+ ```bash
25
+ cd presentations/2026-10-colm-spotlight
26
+ npm install
27
+ node build.mjs
28
+ ```