Spaces:
Running
Running
Merge highlights into results; size for the blog column; pipeline step buttons
Browse files- index.html +31 -10
index.html
CHANGED
|
@@ -401,6 +401,7 @@ pre.code .lang{
|
|
| 401 |
.anim .m{color:var(--dim)} .anim .w{color:var(--hi)} .anim .g{color:var(--acc)}
|
| 402 |
.anim .miss{color:var(--ink)} .anim .hit{color:var(--acc)}
|
| 403 |
.ctl{background:none;border:1px solid var(--rule);color:var(--dim);font:inherit;padding:.1rem 1ch;cursor:pointer}
|
|
|
|
| 404 |
.ctl:hover{color:var(--acc);border-color:var(--acc-dim)}
|
| 405 |
|
| 406 |
footer{margin:6rem 0 0; padding-top:1.5rem; border-top:1px solid var(--rule); color:var(--dim); font-size:.9em}
|
|
@@ -461,7 +462,7 @@ html{scroll-behavior:smooth}
|
|
| 461 |
</div>
|
| 462 |
|
| 463 |
<div class="panel" id="pScaling" hidden>
|
| 464 |
-
<p>Batch encoding matters most in data pipelines, where a tokenizer must process many inputs at once. How well it scales across CPU cores depends on how much state the threads have to share. In
|
| 465 |
<div class="repro" data-repro="threads">
|
| 466 |
<div class="scroll"><div class="chart" id="mtLine"></div></div>
|
| 467 |
<p class="note">Absolute throughput as workers are added, v1 against the released library, on a linear axis from zero.</p>
|
|
@@ -523,7 +524,7 @@ html{scroll-behavior:smooth}
|
|
| 523 |
<div class="key-wins">
|
| 524 |
<article class="key-win">
|
| 525 |
<h3>multilingual by design</h3>
|
| 526 |
-
<p>UTF-8 uses more bytes for many characters outside Latin scripts, and tokenizer implementations can compound that cost with extra splitting work. Performance work often centers English. V1 treats non-Latin languages as part of the performance target.</p>
|
| 527 |
<div class="mini-chart" id="keyLanguageChart"></div>
|
| 528 |
</article>
|
| 529 |
<article class="key-win">
|
|
@@ -549,7 +550,7 @@ html{scroll-behavior:smooth}
|
|
| 549 |
<p>The model stage is where most of the work described here happens. Eight of the ten model families measured in this article use byte pair encoding, or BPE. BPE starts from the bytes of a pre-token and repeatedly joins the highest ranked adjacent pair until no ranked pair remains. The ranking is learned when the tokenizer is trained and ships with it, so the same text always produces the same IDs. A merge never crosses a pre-token boundary. The other two families use WordPiece and Unigram, the two other model types the library supports.</p>
|
| 550 |
<p>The <a href="https://huggingface.co/docs/tokenizers/pipeline">tokenization pipeline</a> page documents the four stages. <a href="https://huggingface.co/docs/transformers/tokenizer_summary">Tokenization algorithms</a> documents BPE, WordPiece and Unigram.</p>
|
| 551 |
<div class="anim" id="a-pipe">
|
| 552 |
-
<div class="cap"><span>one sentence through the pipeline <span class="dim">/ real tokenizer output</span></span><button class="ctl" data-toggle="a-pipe">pause</button></div>
|
| 553 |
<pre class="pipe" id="pipe" aria-label="the five pipeline stages"></pre>
|
| 554 |
<div class="tabs mdl" role="tablist" id="mtabs"></div>
|
| 555 |
<pre class="walk" id="walk"></pre>
|
|
@@ -686,14 +687,25 @@ const reduced = matchMedia("(prefers-reduced-motion: reduce)").matches;
|
|
| 686 |
|
| 687 |
/* ---- animations ------------------------------------------------------- */
|
| 688 |
const running = {};
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 689 |
function toggler(id, step, fps){
|
| 690 |
let t = 0, on = !reduced;
|
| 691 |
running[id] = on;
|
| 692 |
const btn = document.querySelector(`[data-toggle="${id}"]`);
|
| 693 |
-
|
|
|
|
| 694 |
if (btn && reduced) btn.textContent = "play";
|
| 695 |
-
|
| 696 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 697 |
}
|
| 698 |
|
| 699 |
/* ---- decorative rules ------------------------------------------------- */
|
|
@@ -1229,9 +1241,11 @@ renderCrates();
|
|
| 1229 |
};
|
| 1230 |
});
|
| 1231 |
|
| 1232 |
-
toggler
|
| 1233 |
-
|
| 1234 |
-
|
|
|
|
|
|
|
| 1235 |
light(f.box);
|
| 1236 |
$("#walk").innerHTML = `<span class="k">${f.box} · ${f.name}</span>`
|
| 1237 |
+ `<span class="d">${" ".repeat(Math.max(1, 14 - f.name.length))}${esc(f.desc)}</span>\n`
|
|
@@ -1756,8 +1770,15 @@ addEventListener("resize", () => { clearTimeout(rt); rt = setTimeout(drawAll, 12
|
|
| 1756 |
so scrolling="no" cannot be relied on. A few pixels of reflow at an unlucky
|
| 1757 |
width would otherwise raise a scrollbar inside the frame, on top of the
|
| 1758 |
blog's own. HEIGHTS in make_blog_post.py carries the slack that makes
|
| 1759 |
-
hiding the overflow safe rather than lossy.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1760 |
html, body{overflow:hidden}
|
|
|
|
| 1761 |
/* body's background normally propagates to the canvas. Paint it directly so
|
| 1762 |
the slack below the content does not depend on that rule. */
|
| 1763 |
html{background:var(--bg)}
|
|
|
|
| 401 |
.anim .m{color:var(--dim)} .anim .w{color:var(--hi)} .anim .g{color:var(--acc)}
|
| 402 |
.anim .miss{color:var(--ink)} .anim .hit{color:var(--acc)}
|
| 403 |
.ctl{background:none;border:1px solid var(--rule);color:var(--dim);font:inherit;padding:.1rem 1ch;cursor:pointer}
|
| 404 |
+
.ctls{display:flex; gap:.5ch; align-items:center}
|
| 405 |
.ctl:hover{color:var(--acc);border-color:var(--acc-dim)}
|
| 406 |
|
| 407 |
footer{margin:6rem 0 0; padding-top:1.5rem; border-top:1px solid var(--rule); color:var(--dim); font-size:.9em}
|
|
|
|
| 462 |
</div>
|
| 463 |
|
| 464 |
<div class="panel" id="pScaling" hidden>
|
| 465 |
+
<p>Batch encoding matters most in data pipelines, where a tokenizer must process many inputs at once. How well it scales across CPU cores depends on how much state the threads have to share. In a native-thread sweep on Apple M4 Max, v1 scales from one to eight workers at <b>76%</b> of linear.</p>
|
| 466 |
<div class="repro" data-repro="threads">
|
| 467 |
<div class="scroll"><div class="chart" id="mtLine"></div></div>
|
| 468 |
<p class="note">Absolute throughput as workers are added, v1 against the released library, on a linear axis from zero.</p>
|
|
|
|
| 524 |
<div class="key-wins">
|
| 525 |
<article class="key-win">
|
| 526 |
<h3>multilingual by design</h3>
|
| 527 |
+
<p>UTF-8 uses more bytes for many characters outside Latin scripts, and tokenizer implementations can compound that cost with extra splitting work. Performance work often centers on English. V1 treats non-Latin languages as part of the performance target.</p>
|
| 528 |
<div class="mini-chart" id="keyLanguageChart"></div>
|
| 529 |
</article>
|
| 530 |
<article class="key-win">
|
|
|
|
| 550 |
<p>The model stage is where most of the work described here happens. Eight of the ten model families measured in this article use byte pair encoding, or BPE. BPE starts from the bytes of a pre-token and repeatedly joins the highest ranked adjacent pair until no ranked pair remains. The ranking is learned when the tokenizer is trained and ships with it, so the same text always produces the same IDs. A merge never crosses a pre-token boundary. The other two families use WordPiece and Unigram, the two other model types the library supports.</p>
|
| 551 |
<p>The <a href="https://huggingface.co/docs/tokenizers/pipeline">tokenization pipeline</a> page documents the four stages. <a href="https://huggingface.co/docs/transformers/tokenizer_summary">Tokenization algorithms</a> documents BPE, WordPiece and Unigram.</p>
|
| 552 |
<div class="anim" id="a-pipe">
|
| 553 |
+
<div class="cap"><span>one sentence through the pipeline <span class="dim">/ real tokenizer output</span></span><span class="ctls"><button class="ctl" data-step="a-pipe" data-dir="-1" aria-label="previous frame">◀</button><button class="ctl" data-toggle="a-pipe">pause</button><button class="ctl" data-step="a-pipe" data-dir="1" aria-label="next frame">▶</button></span></div>
|
| 554 |
<pre class="pipe" id="pipe" aria-label="the five pipeline stages"></pre>
|
| 555 |
<div class="tabs mdl" role="tablist" id="mtabs"></div>
|
| 556 |
<pre class="walk" id="walk"></pre>
|
|
|
|
| 687 |
|
| 688 |
/* ---- animations ------------------------------------------------------- */
|
| 689 |
const running = {};
|
| 690 |
+
// step(t, dir): t is the running frame count, dir is which way this call moves.
|
| 691 |
+
// dir is 0 on the first paint, 1 for each interval tick, and -1 or 1 when a
|
| 692 |
+
// reader uses the step buttons. An animation that just counts can ignore dir; an
|
| 693 |
+
// animation a reader can walk backwards through needs it, because t alone cannot
|
| 694 |
+
// say whether the last move was forward or back.
|
| 695 |
function toggler(id, step, fps){
|
| 696 |
let t = 0, on = !reduced;
|
| 697 |
running[id] = on;
|
| 698 |
const btn = document.querySelector(`[data-toggle="${id}"]`);
|
| 699 |
+
const label = () => { if (btn) btn.textContent = running[id] ? "pause" : "play"; };
|
| 700 |
+
if (btn) btn.onclick = () => { running[id] = !running[id]; label(); };
|
| 701 |
if (btn && reduced) btn.textContent = "play";
|
| 702 |
+
// Stepping implies wanting to look at a frame, so it pauses first.
|
| 703 |
+
document.querySelectorAll(`[data-step="${id}"]`).forEach(b => {
|
| 704 |
+
const dir = Number(b.dataset.dir) || 1;
|
| 705 |
+
b.onclick = () => { running[id] = false; label(); t += dir; step(t, dir); };
|
| 706 |
+
});
|
| 707 |
+
step(0, 0);
|
| 708 |
+
setInterval(() => { if (running[id]) step(++t, 1); }, 1000 / fps);
|
| 709 |
}
|
| 710 |
|
| 711 |
/* ---- decorative rules ------------------------------------------------- */
|
|
|
|
| 1241 |
};
|
| 1242 |
});
|
| 1243 |
|
| 1244 |
+
// tick is kept here rather than taken from the toggler's counter because the
|
| 1245 |
+
// model tabs reset it: switching model restarts the walk at its first frame.
|
| 1246 |
+
toggler("a-pipe", (t, dir) => {
|
| 1247 |
+
tick = ((tick + dir) % seq.length + seq.length) % seq.length;
|
| 1248 |
+
const f = seq[tick];
|
| 1249 |
light(f.box);
|
| 1250 |
$("#walk").innerHTML = `<span class="k">${f.box} · ${f.name}</span>`
|
| 1251 |
+ `<span class="d">${" ".repeat(Math.max(1, 14 - f.name.length))}${esc(f.desc)}</span>\n`
|
|
|
|
| 1770 |
so scrolling="no" cannot be relied on. A few pixels of reflow at an unlucky
|
| 1771 |
width would otherwise raise a scrollbar inside the frame, on top of the
|
| 1772 |
blog's own. HEIGHTS in make_blog_post.py carries the slack that makes
|
| 1773 |
+
hiding the overflow safe rather than lossy.
|
| 1774 |
+
|
| 1775 |
+
Hiding it unconditionally forced every box to be sized for the worst case,
|
| 1776 |
+
a 320px phone, which left a desktop reader staring at hundreds of pixels of
|
| 1777 |
+
blank. So the boxes are sized for the blog's reading column instead, and
|
| 1778 |
+
below that width the figure scrolls rather than clipping: a scrollbar on a
|
| 1779 |
+
phone is honest, silent truncation is not. */
|
| 1780 |
html, body{overflow:hidden}
|
| 1781 |
+
@media (max-width:640px){ html, body{overflow:auto} }
|
| 1782 |
/* body's background normally propagates to the canvas. Paint it directly so
|
| 1783 |
the slack below the content does not depend on that rule. */
|
| 1784 |
html{background:var(--bg)}
|