Spaces:
Running
Running
Download index.html from lysandre/tokenizers-v1-split: direct link, hf CLI and curl.
- Browser
- Download file 157 kB
-
https://huggingface.co/spaces/lysandre/tokenizers-v1-split/resolve/main/index.html
- Command line
-
hf download hf://spaces/lysandre/tokenizers-v1-split/index.html
-
curl -L -o index.html https://huggingface.co/spaces/lysandre/tokenizers-v1-split/resolve/main/index.html
157 kB
| <html lang="en"> | |
| <head> | |
| <meta charset="utf-8"> | |
| <meta name="viewport" content="width=device-width,initial-scale=1"> | |
| <meta name="description" content="tokenizers v1 encode, decode, scaling and latency, measured with tokbench against tokenizers 0.23."> | |
| <meta property="og:title" content="tokenizers v1 benchmarks"> | |
| <meta property="og:description" content="Tokenizers v1: SOTA"> | |
| </head> | |
| <body data-embed="split"> | |
| <title>tokenizers v1 split</title> | |
| <style> | |
| :root{ | |
| --bg:#000; --ink:#9a9a9a; --hi:#fff; --dim:#7a7a7a; --rule:#1c1c1c; | |
| --acc:#f2c14e; --acc-dim:#7a6229; --ref:#3a3a3a; --acc2:#6fb3aa; | |
| --good:#69b873; --bad:#dc6b6b; | |
| --ch:0.62rem; --lh:1.5; | |
| } | |
| @media (prefers-color-scheme:light){ | |
| :root{ --bg:#fbfbfa; --ink:#4a4a48; --hi:#000; --dim:#6f6f6a; --rule:#e2e2de; | |
| --acc:#8a5a00; --acc-dim:#d8bf8a; --ref:#c4c4c0; --acc2:#2f6f68; | |
| --good:#287a35; --bad:#b33b3b; } | |
| } | |
| :root[data-theme="dark"]{ | |
| --bg:#000; --ink:#9a9a9a; --hi:#fff; --dim:#7a7a7a; --rule:#1c1c1c; | |
| --acc:#f2c14e; --acc-dim:#7a6229; --ref:#3a3a3a; --acc2:#6fb3aa; | |
| --good:#69b873; --bad:#dc6b6b; | |
| } | |
| :root[data-theme="light"]{ | |
| --bg:#fbfbfa; --ink:#4a4a48; --hi:#000; --dim:#6f6f6a; --rule:#e2e2de; | |
| --acc:#8a5a00; --acc-dim:#d8bf8a; --ref:#c4c4c0; --acc2:#2f6f68; | |
| --good:#287a35; --bad:#b33b3b; | |
| } | |
| *{box-sizing:border-box} | |
| html{-webkit-text-size-adjust:100%} | |
| body{ | |
| margin:0; background:var(--bg); color:var(--ink); | |
| font-family:ui-monospace,"SF Mono",SFMono-Regular,Menlo,Consolas,"Liberation Mono",monospace; | |
| font-size:13px; line-height:var(--lh); | |
| font-variant-ligatures:none; font-variant-numeric:tabular-nums; | |
| padding:0 4vw 8rem; | |
| } | |
| @media (max-width:640px){ body{font-size:11px; padding:0 3vw 5rem} } | |
| .wrap{max-width:96ch; margin:0 auto} /* the one measure. Nothing inside caps narrower */ | |
| b,strong{color:var(--hi); font-weight:400} | |
| .hi{color:var(--hi)} | |
| .acc{color:var(--acc)} | |
| .dim{color:var(--dim)} | |
| a{color:var(--hi); text-decoration:none; border-bottom:1px solid var(--rule)} | |
| a:hover{border-bottom-color:var(--acc); color:var(--acc)} | |
| a:focus-visible,button:focus-visible{outline:1px solid var(--acc); outline-offset:2px} | |
| hr{border:0; border-top:1px solid var(--rule); margin:2.5rem 0} | |
| p{margin:0 0 1rem} | |
| pre{margin:0; font:inherit; white-space:pre; overflow-x:auto} | |
| .scroll{overflow-x:auto; overflow-y:hidden} | |
| /* ---------- masthead ---------- */ | |
| header{padding:3.5rem 0 0} | |
| .rule{color:var(--rule); user-select:none; white-space:pre; overflow:hidden} | |
| h1{ | |
| font-size:clamp(1.6rem,7vw,3.4rem); line-height:1.05; margin:1.2rem 0 .4rem; | |
| color:var(--hi); font-weight:400; letter-spacing:-.02em; | |
| } | |
| h1 .v{color:var(--acc)} | |
| .kicker{color:var(--dim); letter-spacing:.22em; text-transform:uppercase; font-size:.78em} | |
| .stand{margin:1.4rem 0 0; color:var(--ink)} | |
| .stand + .stand{margin-top:1rem} | |
| #field{ | |
| color:var(--dim); line-height:1.12; font-size:clamp(6px,1.32vw,12px); | |
| margin:2rem 0 0; white-space:pre; overflow:hidden; user-select:none; | |
| } | |
| #field .t{color:var(--hi)} #field .b{color:var(--acc)} #field .s{color:var(--acc-dim)} | |
| /* ---------- section chrome ---------- */ | |
| section{margin:7rem 0 0; scroll-margin-top:2rem} | |
| h2{ | |
| font-size:1em; font-weight:700; color:var(--hi); margin:0 0 1.4rem; | |
| letter-spacing:.14em; text-transform:uppercase; display:flex; gap:1ch; align-items:baseline; | |
| } | |
| h2 .n{color:var(--acc); letter-spacing:0} | |
| /* 2.1-2.4 are the mechanism behind 02, so they sit one step in */ | |
| section.sub h2{margin-left:3ch} | |
| section.sub h2 .n{color:var(--acc-dim)} | |
| h2 .fill{flex:1; color:var(--rule); overflow:hidden; white-space:nowrap} | |
| /* subheads sit at body size, so weight and the space above carry the hierarchy */ | |
| h3{font-size:1em; font-weight:700; color:var(--hi); margin:4rem 0 1rem; letter-spacing:.04em} | |
| .lede{color:var(--ink); margin:0 0 1.9rem} | |
| /* Not every optimisation reaches every tokenizer family. Say which, up front, | |
| rather than leaving the reader to infer it from the model list. */ | |
| .applies{ | |
| color:var(--acc2); font-size:.82em; letter-spacing:.14em; text-transform:uppercase; | |
| margin:-.7rem 0 1.6rem; | |
| } | |
| .applies::before{content:"[ "; color:var(--rule)} | |
| .applies::after{content:" ]"; color:var(--rule)} | |
| /* ---------- hero stat ---------- */ | |
| .stats{ | |
| display:grid; grid-template-columns:repeat(5,minmax(0,1fr)); gap:0; | |
| width:min(144ch,92vw); margin:3rem 0 0 50%; transform:translateX(-50%); | |
| border-top:1px solid var(--rule); border-bottom:1px solid var(--rule) | |
| } | |
| .stat{padding:1rem 1.4rem; border-right:1px solid var(--rule)} | |
| .stat:first-child{padding-left:0} /* stays aligned with the text column */ | |
| .stat:last-child{border-right:0} | |
| .stat .v{display:block; font-size:clamp(1.5rem,4.2vw,2.4rem); color:var(--hi); line-height:1.1} | |
| .stat.k .v{color:var(--acc)} | |
| .stat .l{display:block; color:var(--dim); font-size:.82em; margin-top:.35rem; letter-spacing:.06em} | |
| @media (max-width:680px){ | |
| .stats{grid-template-columns:repeat(2,minmax(0,1fr))} | |
| .stat:nth-child(even){border-right:0} | |
| .stat:nth-child(n+3){border-top:1px solid var(--rule)} | |
| .stat:nth-child(odd){padding-left:0} | |
| } | |
| @media (max-width:420px){ | |
| .stats{grid-template-columns:1fr} | |
| .stat{border-right:0; border-top:1px solid var(--rule); padding-left:0} | |
| .stat:first-child{border-top:0} | |
| } | |
| /* ---------- three measured wins ---------- */ | |
| .key-wins{ | |
| display:grid; grid-template-columns:repeat(3,minmax(0,1fr)); | |
| width:min(144ch,92vw); margin:0 0 0 50%; transform:translateX(-50%); | |
| border-bottom:1px solid var(--rule) | |
| } | |
| .key-win{min-width:0; padding:2rem 1.4rem 2.2rem; border-right:1px solid var(--rule)} | |
| .key-win:first-child{padding-left:0} | |
| .key-win:last-child{border-right:0} | |
| .key-win h3{margin:0 0 1rem; color:var(--acc); letter-spacing:.02em} | |
| .key-win p{margin:0 0 1.5rem} | |
| .mini-chart{font-size:.86em} | |
| .mini-chart .unit{display:block; color:var(--acc2); margin-bottom:.9rem; line-height:1.4} | |
| .mini-row{display:grid; grid-template-columns:minmax(10ch,auto) 1fr auto; gap:1ch; align-items:center; margin:.55rem 0} | |
| .mini-row .name{color:var(--ink); white-space:nowrap} | |
| .mini-row .track{height:.7rem; background:var(--rule); min-width:3rem} | |
| .mini-row .fill{display:block; height:100%; background:var(--ref)} | |
| .mini-row.us .fill{background:var(--acc)} | |
| .mini-row .value{color:var(--hi); white-space:nowrap} | |
| .mini-row.us .name,.mini-row.us .value{color:var(--acc)} | |
| .key-win .note{margin-top:1rem} | |
| .key-win.repro .repro-panel.on{position:static; width:auto; margin:.8rem 0 0} | |
| @media(max-width:800px){ | |
| .key-wins{grid-template-columns:1fr} | |
| .key-win{border-right:0; border-top:1px solid var(--rule); padding:1.8rem 0} | |
| .key-win:first-child{border-top:0} | |
| } | |
| /* ---------- tabs ---------- */ | |
| .tabs{ | |
| display:flex; flex-wrap:nowrap; gap:2ch; margin:0 0 1.2rem; overflow-x:auto; | |
| scrollbar-width:none; | |
| border-bottom:1px solid var(--rule); padding-bottom:.7rem; | |
| } | |
| .tabs::-webkit-scrollbar{display:none} | |
| #tabs{width:min(160ch,98vw); gap:1.5ch; margin-left:50%; transform:translateX(-50%); font-size:.94em; justify-content:center} | |
| @media(max-width:900px){#tabs{justify-content:flex-start}} | |
| .tabs button{ | |
| flex:0 0 auto; white-space:nowrap; | |
| background:none; border:0; padding:.15rem 0; margin:0; cursor:pointer; color:var(--dim); | |
| font:inherit; letter-spacing:.02em; | |
| } | |
| .tabs button:hover{color:var(--ink)} | |
| .tabs button[aria-selected="true"]{color:var(--acc)} | |
| .tabs button::before{content:"[ "} .tabs button::after{content:" ]"} | |
| .tabs button::before,.tabs button::after{color:var(--rule)} | |
| .tabs button[aria-selected="true"]::before,.tabs button[aria-selected="true"]::after{color:var(--acc-dim)} | |
| @keyframes tab-panel-in{ | |
| from{opacity:0; transform:translateY(4px)} | |
| to{opacity:1; transform:translateY(0)} | |
| } | |
| #ranking .panel.tab-enter{animation:tab-panel-in 160ms ease-out both} | |
| .baseline{display:flex; align-items:center; gap:1ch; margin:0 0 1.2rem; color:var(--dim); font-size:.9em} | |
| .baseline[hidden]{display:none} | |
| .baseline select{ | |
| color:var(--ink); background:var(--bg); border:1px solid var(--rule); border-radius:0; | |
| padding:.25rem .5rem; font:inherit; | |
| } | |
| .mode-help{border-left:1px solid var(--acc-dim); padding-left:1.2ch; margin:0 0 1.6rem; color:var(--dim)} | |
| /* ---------- heading tokenisation ---------- */ | |
| /* headings walk the real pipeline on first sight; the separator is the only | |
| added glyph, so the resting state is exactly the heading text */ | |
| .tk{cursor:pointer} | |
| /* h2 is display-uppercased; token frames must show the tokenizer's real casing, | |
| so the transform is suspended while the animation is running */ | |
| .tk.on{text-transform:none} | |
| .tk .sep{color:var(--acc-dim)} | |
| .tk .id{color:var(--acc)} | |
| h2 .tk,h3 .tk{display:inline} | |
| /* ---------- ascii chart ---------- */ | |
| .chart{white-space:pre; line-height:1.7} | |
| .chart .row{display:block} | |
| .chart .lab{color:var(--ink)} | |
| .chart .lab.us{color:var(--acc)} | |
| .chart .lab.ref{color:var(--dim)} | |
| .chart .bar{color:var(--ref)} | |
| .chart .bar.us{color:var(--acc)} | |
| .chart .bar.hi{color:var(--hi)} | |
| .chart .val{color:var(--hi)} | |
| .chart .val.us{color:var(--acc)} | |
| .chart .x{color:var(--dim)} | |
| .chart .row.inspectable{cursor:pointer} | |
| .chart .row.inspectable .inspect-mark{color:var(--acc2)} | |
| .chart .row.inspectable:hover .lab, | |
| .chart .row.inspectable:hover .val, | |
| .chart .row.inspectable:hover .inspect-mark{color:var(--acc)} | |
| .chart .row.inspectable:focus-visible{outline:1px solid var(--acc); outline-offset:2px} | |
| /* the unit is a caption, not a data row: left-aligned, in the cool accent, with | |
| room so it does not read as another line of prose */ | |
| .chart .unit{color:var(--acc2); margin-bottom:1.4rem; letter-spacing:.06em} | |
| .note{color:var(--dim); font-size:.9em; margin:1.4rem 0 0; line-height:1.6} | |
| /* ---------- measured input preview ---------- */ | |
| .cache-stage{position:relative} | |
| .input-preview{ | |
| border:1px solid var(--rule); border-left-color:var(--acc-dim); | |
| margin:1.2rem 0 0; background:var(--bg); animation:tab-panel-in 160ms ease-out both; | |
| } | |
| .input-preview[hidden]{display:none} | |
| .input-preview-head{ | |
| display:flex; align-items:center; justify-content:space-between; gap:2ch; | |
| padding:.65rem 1ch; border-bottom:1px solid var(--rule); color:var(--acc2); | |
| letter-spacing:.1em; text-transform:uppercase; font-size:.82em; | |
| } | |
| .input-preview-close{ | |
| background:none; border:0; color:var(--dim); font:inherit; cursor:pointer; | |
| padding:.1rem .4rem; | |
| } | |
| .input-preview-close:hover{color:var(--acc)} | |
| .input-segment{padding:.8rem 1ch 0} | |
| .input-segment + .input-segment{border-top:1px solid var(--rule); margin-top:.8rem} | |
| .input-segment-label{color:var(--acc-dim); font-size:.78em; letter-spacing:.08em; text-transform:uppercase; margin-bottom:.45rem} | |
| .input-segment pre{max-height:12rem; overflow:auto; white-space:pre-wrap; word-break:break-word; color:var(--ink); line-height:1.45} | |
| .input-segment.suffix pre{color:var(--hi)} | |
| .input-preview .note{padding:0 1.1ch 1rem; margin-top:.9rem} | |
| @media(min-width:1850px){ | |
| .input-preview{position:absolute; left:calc(100% + 2.5rem); top:0; width:50ch; margin:0} | |
| } | |
| /* ---------- tables ---------- */ | |
| table{border-collapse:collapse; width:100%; margin:1.8rem 0 0; font-size:.95em} | |
| th,td{text-align:right; padding:.34rem 1.1rem .34rem 0; border-bottom:1px solid var(--rule); white-space:nowrap} | |
| th{color:var(--dim); font-weight:400; letter-spacing:.08em; font-size:.85em; text-transform:uppercase} | |
| td:first-child,th:first-child{text-align:left} | |
| /* only the changes table is clickable, so only it gets a hover state */ | |
| #tChanges tbody tr{cursor:pointer} | |
| #tChanges tbody tr:hover td{color:var(--hi)} | |
| #tChanges tbody tr:hover td:first-child a{color:var(--acc); border-bottom-color:var(--acc)} | |
| #tChanges a{border-bottom:1px solid var(--rule)} | |
| #tChanges .go{color:var(--dim); font-size:.85em; margin-left:1ch} | |
| td.k{color:var(--hi)} td.a{color:var(--acc)} td.d{color:var(--dim)} | |
| .unsupported{color:var(--dim); border-bottom:1px dotted var(--dim); cursor:help} | |
| .unsupported:focus{color:var(--hi); outline:1px solid var(--acc-dim); outline-offset:2px} | |
| /* the method table's second column is prose, not a figure. It wraps. */ | |
| #tMethod td:last-child,#tMethod th:last-child, | |
| #tChanges td:nth-child(2),#tChanges th:nth-child(2){white-space:normal; text-align:left; min-width:30ch} | |
| #tChanges td:first-child,#tChanges th:first-child{width:1%} | |
| /* ---------- pipeline diagram ---------- */ | |
| .pipe{line-height:1.5; margin:.4rem 0 1rem; color:var(--dim); overflow-x:auto} | |
| .pipe .s{color:var(--hi)} .pipe .n{color:var(--acc)} .pipe .a{color:var(--acc-dim)} | |
| /* the stage currently doing the work lights up, box and label together */ | |
| /* the stage doing the work is filled solid, so it reads at a glance */ | |
| /* vertical padding on an inline span paints the fill past the glyph without | |
| changing layout, so the three rows of a box join into one solid block */ | |
| .pipe .bx.on{background:var(--acc); padding-block:.27em} | |
| .pipe .bx.on .s{color:var(--bg)} | |
| .pipe .bx.on .e{color:var(--acc)} /* borders vanish into the fill */ | |
| .pipe .num.on{color:var(--acc); font-weight:700} | |
| .crate-result{display:grid; grid-template-columns:1fr auto 1fr; align-items:center; gap:2ch; margin:1.4rem 0} | |
| .crate-size-box{border:1px solid var(--rule); padding:1rem 1.2rem; min-height:5.2rem} | |
| .crate-size-box.current{border-color:var(--acc-dim)} | |
| .crate-size-box .name{display:block; color:var(--dim); margin-bottom:.35rem} | |
| .crate-size-box .size{display:block; color:var(--hi); font-size:1.65em} | |
| .crate-size-box.current .size{color:var(--acc)} | |
| .crate-arrow{color:var(--acc-dim); font-size:1.4em} | |
| .crate-pick{border:1px solid var(--rule); margin:1.2rem 0} | |
| .crate-pick label{display:block; background:var(--bg); padding:.7rem 1ch; cursor:pointer; min-width:0; border-top:1px solid var(--rule)} | |
| .crate-pick>label:first-child,.crate-node>label:first-child{border-top:0} | |
| .crate-pick label.required{cursor:default} | |
| .crate-pick input{accent-color:var(--acc); margin:0 .7ch 0 0} | |
| .crate-pick .cn{color:var(--hi); white-space:nowrap} | |
| .crate-pick .cd{display:block; color:var(--dim); font-size:.82em; margin:.4rem 0 0 2ch; line-height:1.35} | |
| .crate-pick .delta{margin-left:1ch; white-space:nowrap} | |
| .crate-pick .delta.add{color:var(--bad)} | |
| .crate-pick .delta.remove{color:var(--good)} | |
| .crate-features{margin-left:2.2rem; border-left:1px solid var(--acc-dim); background:var(--bg)} | |
| .crate-features label{padding-left:1.2rem} | |
| .crate-profile{display:flex!important; align-items:center; gap:1ch; color:var(--dim)} | |
| .crate-profile select{background:var(--bg); color:var(--hi); border:1px solid var(--rule); padding:.35rem .6rem; font:inherit} | |
| .feature-linked{display:flex; justify-content:space-between; gap:2ch; padding:.8rem 1ch .8rem 1.2rem; border-top:1px solid var(--rule); color:var(--dim)} | |
| .feature-linked strong{color:var(--acc); font-size:1.1em} | |
| @media(max-width:720px){.crate-result{grid-template-columns:1fr}.crate-arrow{transform:rotate(90deg); justify-self:center}.crate-features{margin-left:1rem}} | |
| .tabs.mdl{border-bottom:0; padding-bottom:0; margin:.2rem 0 .9rem; font-size:.9em} | |
| .walk .cur{color:var(--acc)} | |
| .walk{line-height:1.6; overflow-x:auto} | |
| /* the plotted cells must touch vertically to read as a line */ | |
| #mtLine{line-height:1.15} | |
| .walk .k{color:var(--acc)} .walk .d{color:var(--dim)} | |
| .walk .v{color:var(--hi)} .walk .sep{color:var(--acc-dim)} | |
| /* ---------- contents rail ---------- */ | |
| /* Only shown where there is real margin beside the 96ch column; below that it | |
| would either overlap the text or squeeze it. */ | |
| #rail{display:none} | |
| @media (min-width:1180px){ | |
| #rail{ | |
| display:block; position:fixed; top:5rem; left:1.6rem; width:15ch; | |
| font-size:.82em; line-height:1.45; z-index:5; | |
| max-height:calc(100vh - 8rem); overflow-y:auto; | |
| } | |
| #rail ol{list-style:none; margin:0; padding:0; border-left:1px solid var(--rule)} | |
| #rail li{margin:0} | |
| #rail a{ | |
| display:block; padding:.28rem 0 .28rem 1.1ch; color:var(--dim); | |
| border-bottom:0; border-left:1px solid transparent; margin-left:-1px; | |
| text-decoration:none; | |
| } | |
| #rail a:hover{color:var(--ink)} | |
| #rail a .n{color:var(--rule); margin-right:.8ch} | |
| #rail a.sub{padding-left:2.6ch; font-size:.95em} | |
| #rail a.sub .n{margin-right:.6ch} | |
| #rail a[aria-current="true"]{color:var(--acc); border-left-color:var(--acc)} | |
| #rail a[aria-current="true"] .n{color:var(--acc-dim)} | |
| #rail .top{color:var(--dim); letter-spacing:.14em; text-transform:uppercase; | |
| font-size:.85em; padding-bottom:.5rem; display:block} | |
| } | |
| /* ---------- reproduce panel ---------- */ | |
| .repro{position:relative} | |
| .repro-btn{ | |
| background:none; border:1px solid var(--rule); color:var(--dim); font:inherit; | |
| font-size:.85em; padding:.2rem 1ch; margin:1rem 0 0; cursor:pointer; | |
| } | |
| .repro-btn:hover{color:var(--acc); border-color:var(--acc-dim)} | |
| .repro-btn[aria-expanded="true"]{color:var(--acc); border-color:var(--acc-dim)} | |
| .repro-panel{display:none; border:1px solid var(--rule); border-left-color:var(--acc-dim); | |
| padding:.9rem 1.1rem; margin:.8rem 0 0; font-size:.9em} | |
| .repro-panel.on{display:block} | |
| .repro-panel .hd{display:flex; justify-content:space-between; align-items:baseline; | |
| gap:2ch; color:var(--dim); font-size:.85em; letter-spacing:.1em; | |
| text-transform:uppercase; margin-bottom:.6rem} | |
| .repro-panel pre{white-space:pre-wrap; word-break:break-word; color:var(--hi); line-height:1.6} | |
| .repro-panel .cp{background:none; border:1px solid var(--rule); color:var(--dim); | |
| font:inherit; font-size:.9em; padding:.1rem 1ch; cursor:pointer; letter-spacing:0; | |
| text-transform:none} | |
| .repro-panel .cp:hover{color:var(--acc); border-color:var(--acc-dim)} | |
| /* beside the column where there is room, in flow where there is not */ | |
| @media (min-width:1180px){ | |
| .repro-panel.on{position:absolute; left:calc(100% + 2.5rem); top:0; width:36ch; margin:0} | |
| } | |
| /* ---------- code ---------- */ | |
| pre.code{ | |
| border:1px solid var(--rule); border-left:1px solid var(--acc-dim); | |
| padding:.9rem 1.2rem; margin:1.2rem 0; overflow-x:auto; color:var(--hi); | |
| line-height:1.6; position:relative; | |
| } | |
| pre.code .c {color:var(--dim)} /* comment */ | |
| pre.code .s {color:var(--acc2)} /* string */ | |
| pre.code .num{color:var(--acc2)} | |
| pre.code .k {color:var(--acc)} /* keyword */ | |
| pre.code .mac{color:var(--acc)} | |
| pre.code .ty {color:var(--hi); font-weight:700} /* type */ | |
| pre.code .lang{ | |
| display:block; color:var(--dim); font-size:.82em; letter-spacing:.12em; | |
| text-transform:uppercase; margin-bottom:.6rem; | |
| } | |
| /* ---------- roadmap status list ---------- */ | |
| .road{list-style:none; margin:1rem 0 0; padding:0} | |
| /* one flowing column: item names vary from 8 to 26 characters, so a fixed name | |
| column either wraps the long ones or strands the short ones */ | |
| .road li{display:grid; grid-template-columns:2ch 1fr; gap:0 1ch; | |
| padding:.5rem 0; border-bottom:1px solid var(--rule); align-items:baseline} | |
| .road .st{color:var(--dim)} | |
| .road li.done .st{color:var(--acc)} | |
| .road li.wip .st{color:var(--ink)} | |
| .road .nm{color:var(--hi)} | |
| .road li.todo .nm{color:var(--ink)} | |
| .road .dt{color:var(--dim)} | |
| .road .body{line-height:1.65} | |
| .legend{color:var(--dim); font-size:.9em; margin:.6rem 0 0; display:flex; gap:3ch; flex-wrap:wrap} | |
| .legend b{font-weight:400} | |
| .legend .done{color:var(--acc)} .legend .wip{color:var(--ink)} | |
| /* ---------- callout ---------- */ | |
| .box{border-left:1px solid var(--rule); padding:.2rem 0 .2rem 2ch; margin:2.4rem 0; color:var(--ink)} | |
| .box.warn{border-left-color:var(--acc-dim)} | |
| .box .t{color:var(--hi); letter-spacing:.1em; text-transform:uppercase; font-size:.82em; display:block; margin-bottom:.5rem} | |
| /* ---------- animation frames ---------- */ | |
| .anim{border:1px solid var(--rule); padding:1.3rem 1.4rem; margin:2.2rem 0 2.6rem; overflow-x:auto} | |
| .anim .cap{color:var(--dim); font-size:.85em; letter-spacing:.1em; text-transform:uppercase; margin-bottom:.8rem; display:flex; gap:1ch; align-items:baseline; justify-content:space-between} | |
| .anim pre{line-height:1.5} | |
| .anim .m{color:var(--dim)} .anim .w{color:var(--hi)} .anim .g{color:var(--acc)} | |
| .anim .miss{color:var(--ink)} .anim .hit{color:var(--acc)} | |
| .ctl{background:none;border:1px solid var(--rule);color:var(--dim);font:inherit;padding:.1rem 1ch;cursor:pointer} | |
| .ctls{display:flex; gap:.5ch; align-items:center} | |
| .ctl:hover{color:var(--acc);border-color:var(--acc-dim)} | |
| footer{margin:6rem 0 0; padding-top:1.5rem; border-top:1px solid var(--rule); color:var(--dim); font-size:.9em} | |
| html{scroll-behavior:smooth} | |
| @media (prefers-reduced-motion:reduce){ | |
| html{scroll-behavior:auto} | |
| *{animation:none!important; transition:none!important} | |
| } | |
| </style> | |
| <div class="wrap"> | |
| <header> | |
| <div class="rule" id="topRule"></div> | |
| <div class="kicker">Hugging Face · towards a first major version</div> | |
| <h1>tokenizers <span class="v">v1</span></h1> | |
| <p class="stand">The tokenizer has not historically been the bottleneck within ML workflows. Compute-wise, tokenization is light compared to the heavy modeling happening in the rest of the pipeline. Yet, in some cases, it has rapidly become key to accelerating (or slowing down) your machine learning work.</p> | |
| <p class="stand">As models become faster and workloads scale, that balance begins to shift. Training on massive datasets, serving many concurrent requests, or repeatedly processing long inputs can put enough pressure on the tokenizer that it starves the model of data.</p> | |
| <p class="stand">This is why we have chosen to heavily focus on performance for the upcoming version 1 of tokenizers. Tokenization should be light and should scale with your workflow. Your GPUs should never sit idle waiting for the CPU to complete its tokenization.</p> | |
| <p class="stand">In this article, we look at what makes v1 faster than v0.23, often by tens of times.</p> | |
| <p class="stand">This work was entirely possible thanks to the rest of the ecosystem. Tokenization is a very active area of open source work, and libraries such as <a href="https://github.com/marcelroed/gigatoken">gigatoken</a>, <a href="https://crates.io/crates/tiktoken-rs">tiktoken</a>, <a href="https://crates.io/crates/kitoken">kitoken</a>, <a href="https://crates.io/crates/tokie">tokie</a>, <a href="https://crates.io/crates/fastokens">fastokens</a>, <a href="https://crates.io/crates/wordchipper">wordchipper</a> and <a href="https://www.npmjs.com/package/ai-tokenizer">ai-tokenizer</a>, as well as many others, have each pushed on what a fast tokenizer can be. We read that work, and several of the ideas below reached us because another project showed they were worth trying. Before this refactor, tokenizers was nowhere near the performance it could have had, so contributing to it may not have seemed worth it. With this refactor, we hope to make clear that we intend tokenizers to be a library worth contributing to.</p> | |
| <p class="stand">We also thank NVIDIA, IBM and the ExecuTorch team for contributing patches and helping us test across a wide range of hardware to broaden platform support.</p> | |
| <pre id="field" aria-hidden="true"></pre> | |
| </header> | |
| <nav id="rail" aria-label="contents"></nav> | |
| <section id="ranking"> | |
| <h2><span class="n">01</span> <span>results</span> <span class="fill"></span></h2> | |
| <p class="lede">We showcase results for the release candidate of tokenizers v1 against other widely used alternatives. We go over single-threaded, multi-threaded, scaling across threads, per-model comparison, per-language comparison, latency, decoding throughput, memory heap, as well as crate size.</p> | |
| <p class="lede">We run this from the <a href="https://github.com/huggingface/tokbench">tokbench</a> repository, and add a command to rerun the benchmarks on your hardware if you would like to do so.</p> | |
| <div class="tabs" role="tablist" id="tabs"></div> | |
| <label class="baseline" id="scalingModeControl" hidden> | |
| <span>parallelism</span> | |
| <select id="scalingModeSelect"> | |
| <option value="native-threads">native threads</option> | |
| <option value="independent-instances">independent instances</option> | |
| </select> | |
| </label> | |
| <p class="mode-help" id="scalingModeHelp" hidden>Native threads share one tokenizer and let the library distribute a batch across its own worker pool. Independent instances run one tokenizer per worker with no shared state. Tokenizers v1 performs best with native threads, while gigatoken performs best with independent instances.</p> | |
| <div class="panel" id="pBars"> | |
| <div class="repro" data-repro="lead"> | |
| <label class="baseline" id="baselineControl" hidden> | |
| <span>compare v1 with</span> | |
| <select id="baselineSelect"></select> | |
| </label> | |
| <div class="scroll"><div class="chart" id="chart"></div></div> | |
| <p class="note" id="chartNote"></p> | |
| </div> | |
| </div> | |
| <div class="panel" id="pEight" hidden> | |
| <div class="repro" data-repro="threads"> | |
| <div class="scroll"><div class="chart" id="eightChart"></div></div> | |
| <p class="note" id="eightNote"></p> | |
| </div> | |
| </div> | |
| <div class="panel" id="pScaling" hidden> | |
| <p>Batch encoding matters most in data pipelines, where a tokenizer must process many inputs at once. How well it scales across CPU cores depends on how much state the threads have to share. In a native-thread sweep on Apple M4 Max, v1 scales from one to eight workers at <b>76%</b> of linear.</p> | |
| <div class="repro" data-repro="threads"> | |
| <div class="scroll"><div class="chart" id="mtLine"></div></div> | |
| <p class="note">Absolute throughput as workers are added, v1 against the released library, on a linear axis from zero.</p> | |
| </div> | |
| <div class="scroll"><div class="chart" id="mtChart"></div></div> | |
| </div> | |
| <div class="panel" id="pLatency" hidden> | |
| <p>Throughput determines how much text a system can process, while single-encode latency determines how long an individual inference request waits for tokenization.</p> | |
| <div class="repro" data-repro="latency"> | |
| <div class="scroll"><table id="tLat"></table></div> | |
| <p class="note">The time to encode one 512-byte English document with a warm tokenizer, timed call by call. p99 is the slowest 1% of calls.</p> | |
| </div> | |
| </div> | |
| <div class="panel" id="pDecode" hidden> | |
| <p>The reverse path turns token IDs back into text. Across the 6 model families measured for decode, v1 decodes at <b>5.4 to 8.8 times</b> the throughput of tokenizers 0.23 on an M4 Max.</p> | |
| <div class="repro" data-repro="decode"> | |
| <div class="scroll"><div class="chart" id="decodeChart"></div></div> | |
| <p class="note">Decoded UTF-8 output in MB/s.</p> | |
| </div> | |
| </div> | |
| <div class="panel" id="pMemory" hidden> | |
| <p>Runtime memory was measured with the <span class="hi">gpt-oss</span> tokenizer and 1.024 MB of English text on an Apple M4 Max. The selector separates one worker, one tokenizer using eight native threads, and eight independent tokenizer instances.</p> | |
| <label class="baseline"> | |
| <span>configuration</span> | |
| <select id="memoryModeSelect"> | |
| <option value="single">1 worker</option> | |
| <option value="native-threads">8 native threads</option> | |
| <option value="independent-instances">8 independent instances</option> | |
| </select> | |
| </label> | |
| <div class="repro" data-repro="memory"> | |
| <div class="scroll"><table id="tMemory"></table></div> | |
| <p class="note">Live heap after loading the tokenizer and after a warm encode with the returned token buffers released.</p> | |
| </div> | |
| </div> | |
| <div class="panel" id="pCrates" hidden> | |
| <p>Before v1, the Rust implementation exposed encoding, configuration loading, compatibility and training through one crate. V1 divides that implementation into smaller crates. <span class="hi">tk-encode</span> is the required runtime, and applications that depend directly on the subcrates can add <span class="hi">tk-serialize</span>, <span class="hi">tk-convert</span> and <span class="hi">tk-train</span> according to their needs.</p> | |
| <div class="repro" data-repro="crate-size"> | |
| <div class="crate-result" aria-live="polite"> | |
| <div class="crate-size-box"> | |
| <span class="name" id="crateOldName"></span> | |
| <span class="size" id="crateOldSize"></span> | |
| </div> | |
| <span class="crate-arrow" aria-hidden="true">→</span> | |
| <div class="crate-size-box current"> | |
| <span class="name" id="crateCurrentName"></span> | |
| <span class="size" id="crateCurrentSize"></span> | |
| </div> | |
| </div> | |
| <div class="crate-pick" id="cratePick"></div> | |
| </div> | |
| </div> | |
| <div class="stats" id="heroStats"></div> | |
| <div class="key-wins"> | |
| <article class="key-win"> | |
| <h3>multilingual by design</h3> | |
| <p>UTF-8 uses more bytes for many characters outside Latin scripts, and tokenizer implementations can compound that cost with extra splitting work. Performance work often centers on English. V1 treats non-Latin languages as part of the performance target.</p> | |
| <div class="mini-chart" id="keyLanguageChart"></div> | |
| </article> | |
| <article class="key-win"> | |
| <h3>latency reaches the first token</h3> | |
| <p>The teams at Crusoe and NVIDIA make this case especially well in "<a href="https://www.crusoe.ai/resources/blog/reducing-ttft-by-cpumaxxing-tokenization">Reducing TTFT by CPUMaxxing Tokenization</a>." Every prompt must be tokenized before inference can return its first token. That cost becomes visible in time to first token for long agent contexts and requests that reuse a cached model prefix.</p> | |
| <div class="mini-chart" id="keyLatencyChart"></div> | |
| <p class="note">Median p99 across eight model families, measured call by call on an Apple M4 Max. Lower is better.</p> | |
| </article> | |
| <article class="key-win repro" data-repro="architecture"> | |
| <h3>performance across tokenizer families</h3> | |
| <p>The release covers more than BPE. WordPiece gained a double-array trie, a zero-allocation encode path and the shared word cache. Unigram uses the same allocation-conscious pipeline and word cache.</p> | |
| <p>The gains are smaller than for BPE. These two families are where we focus next.</p> | |
| <div class="mini-chart" id="keyArchitectureChart"></div> | |
| <p class="note">BPE: aggregate of headline models; WordPiece uses BERT, Unigram uses T5.</p> | |
| </article> | |
| </div> | |
| </section> | |
| <section id="what"> | |
| <h2><span class="n">02</span> <span>what v1 is</span> <span class="fill"></span></h2> | |
| <p class="lede">v1 will produce the same token IDs as v0.23. The goal was to preserve the output, the API, the vocabulary and the merge ranks, and improve everything that <b>can</b> be improved. That includes breadth. The library stays general across tokenizer families rather than specialising on BPE, so v1 loads everything v0.23 loaded.</p> | |
| <p>A tokenizer converts text into the list of integers a model reads. tokenizers runs that conversion in four stages. Normalization applies operations such as lowercasing or Unicode normalization to the raw text. Pre-tokenization splits the text into smaller pieces called pre-tokens. The model turns each pre-token into tokens and maps them to IDs in its vocabulary. Post-processing adds any special tokens the model expects.</p> | |
| <p>The model stage is where most of the work described here happens. Eight of the ten model families measured in this article use byte pair encoding, or BPE. BPE starts from the bytes of a pre-token and repeatedly joins the highest ranked adjacent pair until no ranked pair remains. The ranking is learned when the tokenizer is trained and ships with it, so the same text always produces the same IDs. A merge never crosses a pre-token boundary. The other two families use WordPiece and Unigram, the two other model types the library supports.</p> | |
| <p>The <a href="https://huggingface.co/docs/tokenizers/pipeline">tokenization pipeline</a> page documents the four stages. <a href="https://huggingface.co/docs/transformers/tokenizer_summary">Tokenization algorithms</a> documents BPE, WordPiece and Unigram.</p> | |
| <div class="anim" id="a-pipe"> | |
| <div class="cap"><span>one sentence through the pipeline <span class="dim">/ real tokenizer output</span></span><span class="ctls"><button class="ctl" data-step="a-pipe" data-dir="-1" aria-label="previous frame">◀</button><button class="ctl" data-toggle="a-pipe">pause</button><button class="ctl" data-step="a-pipe" data-dir="1" aria-label="next frame">▶</button></span></div> | |
| <pre class="pipe" id="pipe" aria-label="the five pipeline stages"></pre> | |
| <div class="tabs mdl" role="tablist" id="mtabs"></div> | |
| <pre class="walk" id="walk"></pre> | |
| </div> | |
| <p>Each stage was worked on. These are the changes that mattered:</p> | |
| <div class="scroll"><table id="tChanges"></table></div> | |
| </section> | |
| <section id="split" class="sub"> | |
| <h2><span class="n">2.1</span> <span>the split: bitstreams instead of a regex</span> <span class="fill"></span></h2> | |
| <p class="applies">applies to most BPE tokenizers</p> | |
| <p class="lede">BPE models use a regular expression to split the input text into smaller, easier to process chunks called pre-tokens. Merges happen inside a pre-token and never across the boundary between two of them, so this split decides what the rest of the pipeline sees.</p> | |
| <p>That regular expression is a fixed parameter of the model. It ships with the tokenizer and never changes at runtime, so there is no need for a general-purpose regex engine to interpret it on every encode. An equivalent splitting function can be written by hand, once, for the pattern a given model actually uses.</p> | |
| <p>A hand-written function can then use the SIMD instructions (single instruction, multiple data) of a modern CPU, which apply one operation to many bytes at once and suit UTF-8 text well. bitcannon views the input's bytes as parallel streams of bits, so boundaries fall out of boolean operations across whole registers instead of a scan that advances one character at a time. It decides 64 bytes per register operation. The same idea drives <a href="https://www.cs.sfu.ca/~ashriram/papers/2012_HPCA_Parabix.pdf">Parabix</a> for text processing and <a href="https://arxiv.org/abs/1902.08318">simdjson</a> for JSON.</p> | |
| <p>This depends on recognising the pattern. A handful of grammars cover most byte-level BPE models, and a tokenizer whose pattern is not among them keeps the regex path and none of this speed-up. That is why the gains in section 01 vary as much as they do.</p> | |
| <div class="anim" id="a-split"> | |
| <div class="cap"><span>split: regex vs bitcannon <span class="dim">/ schematic</span></span><button class="ctl" data-toggle="a-split">pause</button></div> | |
| <pre id="splitPre"></pre> | |
| </div> | |
| <p class="note">The animation illustrates how their work is structured and does not represent timings. Each step advances the regex by one byte and bitcannon by a full register. End-to-end token IDs are verified in section 01. The current public pipeline API does not expose a comparable isolated split timer for both versions, so this section does not assign a speed-up to this stage alone.</p> | |
| </section> | |
| <section id="cache" class="sub"> | |
| <h2><span class="n">2.2</span> <span>the word cache</span> <span class="fill"></span></h2> | |
| <p class="applies">applies to BPE, WordPiece and Unigram</p> | |
| <p class="lede">Real text contains many repeated words. Because BPE always produces the same token IDs for a given pre-token, v1 can save the result after processing it once. A thread-local cache maps each pre-token's bytes to its token IDs, allowing later occurrences to skip the merge process.</p> | |
| <p>Naturally, as the input grows, the number of unique words can grow more slowly than the total number of words. Repeated words then account for an increasing share of the input. New words still appear, which accounts for the occasional misses in the animation below.</p> | |
| <div class="anim" id="a-cache"> | |
| <div class="cap"><span>cache stream: what a hit actually saves <span class="dim">/ schematic</span></span><button class="ctl" data-toggle="a-cache">pause</button></div> | |
| <pre id="cachePre"></pre> | |
| </div> | |
| <p class="note" style="margin:0 0 1rem">The bar represents the work required to convert each pre-token into token IDs. A cache miss runs the full BPE merge loop, so the bar fills slowly. A cache hit requires only a lookup and finishes sooner. The animation is schematic.</p> | |
| <div id="cacheResult" class="cache-stage" data-inspect-label="inspect request" | |
| data-sample-label="opening excerpt" | |
| data-prefix-label="shared 8 KiB prefix, opening excerpt" | |
| data-suffix-label="unique 2 KiB suffix, opening excerpt" | |
| data-corpus-note="The benchmark measures the complete corpus. The screen shows an opening excerpt from the measured input." | |
| data-prefix-note="The benchmark measures the complete request. The screen shows opening excerpts from the shared prefix and this request's unique suffix." hidden> | |
| <div class="repro" data-repro="cache"> | |
| <div class="scroll"><div class="chart" id="cacheChart"></div></div> | |
| <aside id="cachePreview" class="input-preview" aria-labelledby="cachePreviewTitle" hidden> | |
| <div class="input-preview-head"> | |
| <span id="cachePreviewTitle">measured input</span> | |
| <button class="input-preview-close" type="button">close</button> | |
| </div> | |
| <div id="cachePreviewSegments"></div> | |
| <p class="note" id="cachePreviewNote"></p> | |
| </aside> | |
| <p class="note">Cache enabled divided by cache disabled, measured on an Apple M4 Max. Corpus bars are medians across 8 model families in one complete report. Shared prompt prefix is the median across 3 reports and 8 model cells per report. Each report uses 100 distinct 10 KiB requests from a real agent trace, with a shared 8 KiB prefix and a different suffix. Both configurations produced the same token IDs.</p> | |
| <p class="note">Reproduce the shared-prefix result with <span class="hi">tokbench measure prefix-sharing --engine pipeline --engine hf-tokenizers --compare-to pipeline-no-cache --corpus agentic_swe</span>.</p> | |
| </div> | |
| </div> | |
| <div class="box warn"><span class="t">caveat</span> | |
| Caching works best when the input contains repeated pre-tokens. Input with few repeated pre-tokens can pay for lookups without receiving many hits.</div> | |
| </section> | |
| <section id="merge" class="sub"> | |
| <h2><span class="n">2.3</span> <span>the merge loop</span> <span class="fill"></span></h2> | |
| <p class="applies">applies to BPE only</p> | |
| <p class="lede">The next major cost comes from the BPE merge loop. For each pre-token, the loop repeatedly finds the highest-priority adjacent pair and merges it. The previous implementation allocated new memory for every call and built a new priority queue for every pre-token.</p> | |
| <p>v1 reuses a scratch buffer owned by the caller, removing those repeated allocations. It stores symbols in a flat array and links adjacent symbols by their positions in that array, which makes updates during merging cheaper. It also processes a batch of pre-tokens in a single model call.</p> | |
| <p>Each candidate pair is also packed into a single 64-bit value, with the merge rank in the high bits. Comparing two candidates is then just comparing two integers, and "no merge here" is the largest possible value, so the loop finds its next merge without a branch.</p> | |
| </section> | |
| <section id="method"> | |
| <h2><span class="n">03</span> <span>method</span> <span class="fill"></span></h2> | |
| <p class="lede">Small differences in benchmark design can produce large differences in tokenizer performance. We used the following rules to keep the comparison consistent across engines.</p> | |
| <div class="scroll"><table id="tMethod"></table></div> | |
| <div class="box"><span class="t">the measurement regime dominates</span> | |
| Repeatedly encoding one document can be faster than encoding a stream of distinct documents on the same build. The first approach measures performance when the entire document is already represented in the cache. The second measures performance on new input while allowing previously seen pre-tokens to remain cached.</p> | |
| <p>Both conditions are sometimes described as "warm," even though they measure different workloads. Our headline results use distinct documents, and the complete corpus is too large to fit in the cache. Tokenizer benchmarks should identify which workload they use because the choice can dominate the result.</div> | |
| </section> | |
| <section id="end"> | |
| <h2><span class="n">04</span> <span>what this adds up to</span> <span class="fill"></span></h2> | |
| <p class="lede">Across the ten model families v1's encode path covers, it encodes text <b>3 to 30 times faster</b> than v0.23 with one thread on an Apple M4 Max. The low end is t5-base, the high end gpt2. It scales at <b>76%</b> of linear across eight workers. Throughout these changes, v1 produces exactly the same token IDs as the released library.</p> | |
| <p>The overall improvement comes from several changes working together: a hand-written splitter in place of a regex engine, a cache that answers a repeated word without merging it again, a merge loop that never touches the allocator, and one model call per batch of pre-tokens instead of one per pre-token. Each reduces the work done at a different point in the pipeline.</p> | |
| <p>The next priority is support for more model families. We will move additional models onto the new merge loop before <span class="hi">1.0.0</span>. The following section tracks that work.</p> | |
| <p>This page is generated from <a href="https://github.com/huggingface/tokbench">tokbench</a> results and will be updated as support expands.</p> | |
| </section> | |
| <section id="start"> | |
| <h2><span class="n">05</span> <span>getting it</span> <span class="fill"></span></h2> | |
| <p class="lede">A release candidate for v1 is on crates.io. The API you call is the one you already call, so the only thing that changes is which build you install.</p> | |
| <p></p><p>It is the ordinary install:</p><pre class="code"><span class="lang">bash</span>cargo add tokenizers --pre | |
| </pre><p>Training is behind a default-on feature that pulls a C++ dependency with it. If you only need to encode, turn it off to exclude the training implementation:</p><pre class="code"><span class="lang">bash</span>cargo add tokenizers --pre --no-default-features --features http | |
| </pre><p>Encoding is unchanged: same call, same ids.</p><pre class="code"><span class="lang">rust</span><span class="k">use</span> tokenizers::tokenizer::{<span class="ty">Result</span>, <span class="ty">Tokenizer</span>}; | |
| <span class="k">fn</span> main() -> <span class="ty">Result</span><()> { | |
| <span class="k">let</span> tokenizer = <span class="ty">Tokenizer</span>::from_pretrained(<span class="s">"deepseek-ai/DeepSeek-V4-Flash"</span>, <span class="k">None</span>)?; | |
| <span class="k">let</span> encoding = tokenizer.encode(<span class="s">"The tokenizer is no longer the bottleneck."</span>, <span class="k">false</span>)?; | |
| <span class="mac">println!</span>(<span class="s">"{:?}"</span>, encoding.get_ids()); | |
| <span class="c">// [671, 17840, 9160, 344, 1119, 5827, 270, 111127, 16]</span> | |
| <span class="mac">println!</span>(<span class="s">"{:?}"</span>, encoding.get_tokens()); | |
| <span class="c">// ["The", "Ġtoken", "izer", "Ġis", "Ġno", "Ġlonger", "Ġthe", "Ġbottleneck", "."]</span> | |
| <span class="k">Ok</span>(()) | |
| } | |
| </pre><p>For a batch, <span class="hi">encode_batch</span> is what scales across cores. It is the call the threads section measures.</p><pre class="code"><span class="lang">rust</span><span class="k">let</span> encodings = tokenizer.encode_batch(documents, <span class="k">false</span>)?; | |
| </pre><p>Every figure on this page was measured against this crate. The Python bindings wrap the same code and are built from <span class="hi">bindings/python</span>, but they add per-call overhead that none of these measurements include.</p><p></p> | |
| </section> | |
| <section id="road"> | |
| <h2><span class="n">06</span> <span>progress towards v1</span> <span class="fill"></span></h2> | |
| <p class="lede">The benchmarks on this page cover the completed release-candidate work listed first. The remaining sections show what is still required for <span class="hi">1.0.0</span> and what we plan to explore afterward.</p> | |
| <p class="legend"><span><span class="done">█</span> shipped</span> | |
| <span><span class="wip">▓</span> in progress</span> | |
| <span><span>·</span> planned</span></p> | |
| <h3>release candidate: implemented</h3> | |
| <p class="note" style="margin:0 0 .4rem">This work is in the Rust pre-release on crates.io. Install it with <span class="hi">cargo add tokenizers --pre</span>.</p> | |
| <ul class="road"><li class="done"><span class="st">█</span><span class="body">workspace split: divide the single crate into <span class="hi">tk-encode</span>, <span class="hi">tk-serialize</span>, <span class="hi">tk-convert</span> and <span class="hi">tk-train</span>, so an application links only what it uses</span></li><li class="done"><span class="st">█</span><span class="body">bitcannon: replace regex splitting on the encoding path with bitstream operations covering GPT-2, cl100k, o200k, Tekken and DeepSeek. This replaced the finite-state machines that shipped first <a href="https://github.com/huggingface/tokenizers/pull/2201">#2201</a> <a href="https://github.com/huggingface/tokenizers/pull/2317">#2317</a></span></li><li class="done"><span class="st">█</span><span class="body">WordCache: reuse the token IDs of previously processed pre-tokens <a href="https://github.com/huggingface/tokenizers/pull/2262">#2262</a>, <span class="hi">af5a3e3</span></span></li><li class="done"><span class="st">█</span><span class="body">faster lookup and merging structures: add FlatCache, MPHF RankStore, incremental merging, and BucketVocabStore <a href="https://github.com/huggingface/tokenizers/pull/2190">#2190</a> <a href="https://github.com/huggingface/tokenizers/pull/2188">#2188</a></span></li><li class="done"><span class="st">█</span><span class="body">reusable model memory: move temporary model state into scratch buffers so tokenization does not allocate on each call <a href="https://github.com/huggingface/tokenizers/pull/2175">#2175</a> <a href="https://github.com/huggingface/tokenizers/pull/2183">#2183</a></span></li><li class="done"><span class="st">█</span><span class="body">pipeline post-processing: expose post-processing as the <span class="hi">STAGE_POST</span> pipeline stage <a href="https://github.com/huggingface/tokenizers/pull/2182">#2182</a></span></li><li class="done"><span class="st">█</span><span class="body">batched model calls: process multiple pre-token spans in one call <a href="https://github.com/huggingface/tokenizers/pull/2304">#2304</a></span></li><li class="done"><span class="st">█</span><span class="body">faster decoding: write decoded bytes directly into a reusable buffer, avoid intermediate strings and copies, accelerate token lookup, support buffered streaming, and decode batches in parallel</span></li><li class="done"><span class="st">█</span><span class="body"><span class="hi">role_to_token</span> support <a href="https://github.com/huggingface/tokenizers/pull/2343">#2343</a></span></li><li class="done"><span class="st">█</span><span class="body">Node.js bindings <a href="https://github.com/huggingface/tokenizers/pull/2281">#2281</a></span></li></ul> | |
| <h3>1.0.0</h3> | |
| <ul class="road"><li class="todo"><span class="st">·</span><span class="body">one encoding implementation: use <span class="hi">tk-encode</span> during training validation so training and inference cannot produce different tokenization results</span></li><li class="todo"><span class="st">·</span><span class="body">optional offsets and masks: compute this metadata only when requested, keeping it off the token-ID-only path</span></li><li class="todo"><span class="st">·</span><span class="body">rework normalizers</span></li><li class="todo"><span class="st">·</span><span class="body">bitnorm support, building on atomnorm <a href="https://github.com/huggingface/tokenizers/pull/2209">#2209</a></span></li><li class="todo"><span class="st">·</span><span class="body">spm precompiled</span></li><li class="todo"><span class="st">·</span><span class="body">simpler Python bindings: reduce locking, wrapper types, and handwritten dispatch code while preserving subclassing, serialization, custom decoders, mutation behavior, and support for free-threaded CPython</span></li><li class="todo"><span class="st">·</span><span class="body">inference-only C and C++ bindings for ExecuTorch and llama.cpp, with possible JVM, Swift, and Go bindings to follow</span></li></ul> | |
| <h3>after 1.0.0</h3> | |
| <ul class="road"><li class="todo"><span class="st">·</span><span class="body">tok-devices: explore GPU encoding and batch decoding while keeping text and token IDs on the device. The decoder would upload the vocabulary once, calculate output positions in parallel, and gather the corresponding bytes on the GPU. This would be an optional component intended for large batches, subject to further prototyping and measurement.</span></li></ul> | |
| </section> | |
| </div> | |
| <script id="data" type="application/json">{"tabs":[{"id":"single","label":"single thread","unit":"MB/s","rows":[{"name":"tokenizers v1","v":139.6,"x":15.9,"us":true,"ref":false},{"name":"gigatoken","v":137.2,"x":15.6,"us":false,"ref":false},{"name":"fastokens","v":59.6,"x":6.8,"us":false,"ref":false},{"name":"wordchipper","v":50.1,"x":5.7,"us":false,"ref":false},{"name":"tiktoken","v":29.3,"x":3.3,"us":false,"ref":false},{"name":"tokie","v":26.7,"x":3.0,"us":false,"ref":false},{"name":"kitoken","v":25.8,"x":2.9,"us":false,"ref":false},{"name":"tokenizers 0.23","v":8.8,"x":1.0,"us":false,"ref":true}],"note":""},{"id":"mt","label":"8 threads","panel":"pEight","unit":"MB/s","rows":[{"name":"tokenizers v1","v":833,"us":true,"ref":false,"curve":[130,245,450,833],"eff":76,"range":[58,98],"n":16,"x":18.9},{"name":"fastokens","v":166,"us":false,"ref":false,"curve":[153,100,139,166],"eff":13,"range":[8,15],"n":14,"x":3.8},{"name":"gigatoken","v":101,"us":false,"ref":false,"curve":[177,106,103,101],"eff":16,"range":[4,24],"n":16,"x":2.3},{"name":"tokenizers 0.23","v":44,"us":false,"ref":true,"curve":[8,15,26,44],"eff":77,"range":[63,85],"n":16,"x":1.0}],"threads":[1,2,4,8],"note":""},{"id":"eff","label":"scaling","panel":"pScaling","unit":"% of linear","rows":[{"name":"tokenizers 0.23","v":77,"us":false,"ref":true,"curve":[8,15,26,44],"eff":77,"range":[63,85],"n":16,"x":null},{"name":"tokenizers v1","v":76,"us":true,"ref":false,"curve":[130,245,450,833],"eff":76,"range":[58,98],"n":16,"x":null},{"name":"gigatoken","v":16,"us":false,"ref":false,"curve":[177,106,103,101],"eff":16,"range":[4,24],"n":16,"x":null},{"name":"fastokens","v":13,"us":false,"ref":false,"curve":[153,100,139,166],"eff":13,"range":[8,15],"n":14,"x":null}],"max":100,"dec":0,"note":""},{"id":"models","label":"by model","pivot":1,"default_baseline":"hf-tokenizers","comparisons":{"hf-tokenizers":{"label":"tokenizers 0.23","unit":"× v1 / tokenizers 0.23","rows":[{"name":"gpt2","v":29.83,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"llama-3","v":26.58,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"glm-5.2","v":24.19,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"deepseek-v4","v":14.63,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"qwen2","v":13.0,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"minimax","v":12.42,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"gpt-oss","v":11.69,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"nemotron-3","v":10.74,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"bert-base-uncased","v":6.73,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"t5-base","v":3.33,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"}],"matched":220,"eligible":220,"note":""},"fastokens":{"label":"fastokens","unit":"× v1 / fastokens","rows":[{"name":"llama-3","v":3.66,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"glm-5.2","v":3.12,"x":null,"us":true,"ref":false,"n":21,"count_unit":"corpora"},{"name":"deepseek-v4","v":2.78,"x":null,"us":true,"ref":false,"n":21,"count_unit":"corpora"},{"name":"nemotron-3","v":2.01,"x":null,"us":true,"ref":false,"n":21,"count_unit":"corpora"},{"name":"qwen2","v":1.89,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"gpt-oss","v":1.83,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"minimax","v":1.66,"x":null,"us":true,"ref":false,"n":21,"count_unit":"corpora"}],"matched":150,"eligible":176,"note":""},"gigatoken":{"label":"gigatoken","unit":"× v1 / gigatoken","rows":[{"name":"deepseek-v4","v":1.22,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"gpt-oss","v":1.06,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"glm-5.2","v":1.03,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"llama-3","v":0.99,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"gpt2","v":0.98,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"nemotron-3","v":0.94,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"qwen2","v":0.94,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"minimax","v":0.93,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"}],"matched":176,"eligible":176,"note":""},"kitoken":{"label":"kitoken","unit":"× v1 / kitoken","rows":[{"name":"llama-3","v":6.94,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"glm-5.2","v":6.9,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"gpt2","v":6.36,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"deepseek-v4","v":5.75,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"qwen2","v":5.03,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"minimax","v":4.8,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"gpt-oss","v":4.13,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"nemotron-3","v":3.67,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"}],"matched":176,"eligible":176,"note":""},"tiktoken":{"label":"tiktoken","unit":"× v1 / tiktoken","rows":[{"name":"gpt2","v":9.74,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"llama-3","v":6.3,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"glm-5.2","v":5.7,"x":null,"us":true,"ref":false,"n":21,"count_unit":"corpora"},{"name":"qwen2","v":3.91,"x":null,"us":true,"ref":false,"n":20,"count_unit":"corpora"},{"name":"gpt-oss","v":3.79,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"minimax","v":3.73,"x":null,"us":true,"ref":false,"n":19,"count_unit":"corpora"},{"name":"nemotron-3","v":3.34,"x":null,"us":true,"ref":false,"n":21,"count_unit":"corpora"}],"matched":147,"eligible":176,"note":""},"tokie":{"label":"tokie","unit":"× v1 / tokie","rows":[{"name":"gpt-oss","v":6.26,"x":null,"us":true,"ref":false,"n":15,"count_unit":"corpora"},{"name":"llama-3","v":5.43,"x":null,"us":true,"ref":false,"n":15,"count_unit":"corpora"},{"name":"glm-5.2","v":5.37,"x":null,"us":true,"ref":false,"n":17,"count_unit":"corpora"},{"name":"nemotron-3","v":4.74,"x":null,"us":true,"ref":false,"n":12,"count_unit":"corpora"},{"name":"minimax","v":3.83,"x":null,"us":true,"ref":false,"n":17,"count_unit":"corpora"},{"name":"deepseek-v4","v":3.63,"x":null,"us":true,"ref":false,"n":18,"count_unit":"corpora"},{"name":"gpt2","v":3.52,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"qwen2","v":2.7,"x":null,"us":true,"ref":false,"n":18,"count_unit":"corpora"}],"matched":134,"eligible":176,"note":""},"wordchipper":{"label":"wordchipper","unit":"× v1 / wordchipper","rows":[{"name":"qwen2","v":5.42,"x":null,"us":true,"ref":false,"n":20,"count_unit":"corpora"},{"name":"nemotron-3","v":4.21,"x":null,"us":true,"ref":false,"n":21,"count_unit":"corpora"},{"name":"gpt2","v":3.6,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"llama-3","v":3.09,"x":null,"us":true,"ref":false,"n":16,"count_unit":"corpora"},{"name":"glm-5.2","v":2.97,"x":null,"us":true,"ref":false,"n":12,"count_unit":"corpora"},{"name":"minimax","v":2.01,"x":null,"us":true,"ref":false,"n":19,"count_unit":"corpora"},{"name":"gpt-oss","v":1.89,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"}],"matched":132,"eligible":176,"note":""}},"unit":"× v1 / tokenizers 0.23","rows":[{"name":"gpt2","v":29.83,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"llama-3","v":26.58,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"glm-5.2","v":24.19,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"deepseek-v4","v":14.63,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"qwen2","v":13.0,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"minimax","v":12.42,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"gpt-oss","v":11.69,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"nemotron-3","v":10.74,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"bert-base-uncased","v":6.73,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"},{"name":"t5-base","v":3.33,"x":null,"us":true,"ref":false,"n":22,"count_unit":"corpora"}],"note":""},{"id":"languages","label":"by language","pivot":1,"default_baseline":"hf-tokenizers","comparisons":{"hf-tokenizers":{"label":"tokenizers 0.23","unit":"× v1 / tokenizers 0.23","rows":[{"name":"English","v":22.44,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Amharic","v":14.59,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Hindi","v":12.52,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Hebrew","v":12.07,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Bengali","v":11.9,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Arabic","v":11.28,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Greek","v":11.11,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Georgian","v":10.2,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Korean","v":9.34,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Tamil","v":7.68,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Japanese","v":7.11,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Russian","v":7.07,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Chinese","v":6.77,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Thai","v":6.08,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"}],"matched":140,"eligible":140,"note":""},"fastokens":{"label":"fastokens","unit":"× v1 / fastokens","rows":[{"name":"English","v":3.56,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Bengali","v":2.65,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Hindi","v":2.34,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Amharic","v":2.01,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Tamil","v":1.94,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Georgian","v":1.91,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Arabic","v":1.9,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Thai","v":1.7,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Greek","v":1.67,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Hebrew","v":1.65,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Japanese","v":1.2,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Korean","v":1.13,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Russian","v":1.08,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Chinese","v":1.02,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"}],"matched":98,"eligible":112,"note":""},"gigatoken":{"label":"gigatoken","unit":"× v1 / gigatoken","rows":[{"name":"Japanese","v":1.65,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Thai","v":1.53,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Chinese","v":1.46,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Bengali","v":1.42,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Tamil","v":1.36,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Hindi","v":1.35,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Korean","v":1.17,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Georgian","v":1.16,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Greek","v":1.1,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Arabic","v":1.05,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Hebrew","v":1.02,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Amharic","v":0.99,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Russian","v":0.91,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"English","v":0.77,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"}],"matched":112,"eligible":112,"note":""},"kitoken":{"label":"kitoken","unit":"× v1 / kitoken","rows":[{"name":"English","v":5.8,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Hindi","v":5.76,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Bengali","v":5.24,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Hebrew","v":4.33,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Greek","v":4.25,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Georgian","v":4.2,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Amharic","v":4.18,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Tamil","v":4.18,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Thai","v":4.05,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Arabic","v":3.96,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Korean","v":3.42,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Japanese","v":3.27,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Chinese","v":2.77,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Russian","v":2.56,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"}],"matched":112,"eligible":112,"note":""},"tiktoken":{"label":"tiktoken","unit":"× v1 / tiktoken","rows":[{"name":"English","v":5.34,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Bengali","v":4.28,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Tamil","v":4.16,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Amharic","v":4.05,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Hindi","v":4.03,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Thai","v":3.8,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Georgian","v":3.73,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Arabic","v":3.68,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Hebrew","v":3.54,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Greek","v":3.5,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Japanese","v":2.91,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Korean","v":2.7,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Chinese","v":2.37,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Russian","v":2.23,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"}],"matched":98,"eligible":112,"note":""},"tokie":{"label":"tokie","unit":"× v1 / tokie","rows":[{"name":"Tamil","v":9.15,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Georgian","v":8.62,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Bengali","v":7.38,"x":null,"us":true,"ref":false,"n":6,"count_unit":"models"},{"name":"Thai","v":6.79,"x":null,"us":true,"ref":false,"n":5,"count_unit":"models"},{"name":"Hindi","v":6.5,"x":null,"us":true,"ref":false,"n":6,"count_unit":"models"},{"name":"Amharic","v":5.35,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Japanese","v":5.03,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Chinese","v":4.56,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Arabic","v":4.34,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Hebrew","v":4.21,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Greek","v":4.19,"x":null,"us":true,"ref":false,"n":8,"count_unit":"models"},{"name":"Russian","v":3.13,"x":null,"us":true,"ref":false,"n":6,"count_unit":"models"},{"name":"Korean","v":2.83,"x":null,"us":true,"ref":false,"n":1,"count_unit":"models"},{"name":"English","v":1.46,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"}],"matched":94,"eligible":112,"note":""},"wordchipper":{"label":"wordchipper","unit":"× v1 / wordchipper","rows":[{"name":"Tamil","v":4.82,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Bengali","v":4.69,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Hindi","v":4.21,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Amharic","v":3.3,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Arabic","v":3.08,"x":null,"us":true,"ref":false,"n":5,"count_unit":"models"},{"name":"Korean","v":3.04,"x":null,"us":true,"ref":false,"n":5,"count_unit":"models"},{"name":"Russian","v":2.48,"x":null,"us":true,"ref":false,"n":5,"count_unit":"models"},{"name":"Greek","v":2.38,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Georgian","v":2.35,"x":null,"us":true,"ref":false,"n":5,"count_unit":"models"},{"name":"Hebrew","v":2.33,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Chinese","v":2.27,"x":null,"us":true,"ref":false,"n":6,"count_unit":"models"},{"name":"English","v":2.21,"x":null,"us":true,"ref":false,"n":7,"count_unit":"models"},{"name":"Japanese","v":2.0,"x":null,"us":true,"ref":false,"n":5,"count_unit":"models"},{"name":"Thai","v":1.82,"x":null,"us":true,"ref":false,"n":6,"count_unit":"models"}],"matched":86,"eligible":112,"note":""}},"unit":"× v1 / tokenizers 0.23","rows":[{"name":"English","v":22.44,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Amharic","v":14.59,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Hindi","v":12.52,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Hebrew","v":12.07,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Bengali","v":11.9,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Arabic","v":11.28,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Greek","v":11.11,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Georgian","v":10.2,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Korean","v":9.34,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Tamil","v":7.68,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Japanese","v":7.11,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Russian","v":7.07,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Chinese","v":6.77,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Thai","v":6.08,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"}],"note":""},{"id":"lat","label":"latency","panel":"pLatency"},{"id":"decode","label":"decode","panel":"pDecode"},{"id":"memory","label":"memory","panel":"pMemory"},{"id":"crates","label":"crate size","panel":"pCrates"}],"meta":{"lat_lo":9,"lat_hi":28,"decode_platform":"an M4 Max","decode_lo":5.4,"decode_hi":8.8,"decode_reports":1,"decode_cells":130,"decode_models":6,"decode_engines":5,"memory_input_mb":"1.024","memory_reps":3,"memory_loaded_ratio":2.8,"slimest_reduction":4.1,"cache_reports":1,"cache_cells":48,"cache_models":8,"prefix_cache_reports":3,"prefix_cache_cells":8,"prefix_shared_kib":8,"prefix_request_kib":10,"prefix_requests":100,"eff_v1":76,"eff_v1_lo":58,"eff_v1_hi":98,"eff_ref":77,"eff_ref_lo":63,"eff_ref_hi":85,"eff_v1_claim":"The observed range includes or falls below 100%, so we describe this as roughly linear.","models":["deepseek-v4","glm-5.2","gpt-oss","gpt2","llama-3","minimax","nemotron-3","qwen2"],"rev":"tokenizers-rc0 @ 199d9a13","hero_x":15.9,"hero_lo":3,"hero_hi":30,"hero_lo_model":"t5-base","hero_hi_model":"gpt2","verified":176,"cells":176},"scaling_modes":{"native-threads":{"mt_all":{"hf-tokenizers":[{"threads":1,"mbps":7.5228663600034,"efficiency_pct":100.0},{"threads":2,"mbps":14.521385608675022,"efficiency_pct":96.7938955950852},{"threads":4,"mbps":26.020517285900514,"efficiency_pct":88.19712617803701},{"threads":8,"mbps":44.0855326530704,"efficiency_pct":77.22426538734213}],"pipeline":[{"threads":1,"mbps":130.19232277600958,"efficiency_pct":100.0},{"threads":2,"mbps":245.10005801837784,"efficiency_pct":95.67243108746871},{"threads":4,"mbps":450.4032565484989,"efficiency_pct":86.24868422308896},{"threads":8,"mbps":832.7926976264296,"efficiency_pct":76.08074670702482}],"fastokens":[{"threads":1,"mbps":153.13629691139056,"efficiency_pct":100.0},{"threads":2,"mbps":100.18596059365922,"efficiency_pct":32.46921028590421},{"threads":4,"mbps":139.41812998827487,"efficiency_pct":22.831554872149084},{"threads":8,"mbps":165.50227756173405,"efficiency_pct":12.974924493501378}],"gigatoken":[{"threads":1,"mbps":176.58144684171077,"efficiency_pct":100.0},{"threads":2,"mbps":105.61840529054237,"efficiency_pct":64.21142226511743},{"threads":4,"mbps":102.98862250812206,"efficiency_pct":31.49840382010632},{"threads":8,"mbps":101.1271577299603,"efficiency_pct":15.87513924631837}]},"mt":{"id":"mt","label":"8 threads","panel":"pEight","unit":"MB/s","rows":[{"name":"tokenizers v1","v":833,"us":true,"ref":false,"curve":[130,245,450,833],"eff":76,"range":[58,98],"n":16,"x":18.9},{"name":"fastokens","v":166,"us":false,"ref":false,"curve":[153,100,139,166],"eff":13,"range":[8,15],"n":14,"x":3.8},{"name":"gigatoken","v":101,"us":false,"ref":false,"curve":[177,106,103,101],"eff":16,"range":[4,24],"n":16,"x":2.3},{"name":"tokenizers 0.23","v":44,"us":false,"ref":true,"curve":[8,15,26,44],"eff":77,"range":[63,85],"n":16,"x":1.0}],"threads":[1,2,4,8],"note":""},"eff":{"id":"eff","label":"scaling","panel":"pScaling","unit":"% of linear","rows":[{"name":"tokenizers 0.23","v":77,"us":false,"ref":true,"curve":[8,15,26,44],"eff":77,"range":[63,85],"n":16,"x":null},{"name":"tokenizers v1","v":76,"us":true,"ref":false,"curve":[130,245,450,833],"eff":76,"range":[58,98],"n":16,"x":null},{"name":"gigatoken","v":16,"us":false,"ref":false,"curve":[177,106,103,101],"eff":16,"range":[4,24],"n":16,"x":null},{"name":"fastokens","v":13,"us":false,"ref":false,"curve":[153,100,139,166],"eff":13,"range":[8,15],"n":14,"x":null}],"max":100,"dec":0,"note":""}},"independent-instances":{"mt_all":{"hf-tokenizers":[{"threads":1,"mbps":7.6739991535884595,"efficiency_pct":100.0},{"threads":2,"mbps":14.608239992088775,"efficiency_pct":94.15586748013827},{"threads":4,"mbps":25.987505233670404,"efficiency_pct":84.40910799756492},{"threads":8,"mbps":43.685177194923426,"efficiency_pct":72.54628446542785}],"pipeline":[{"threads":1,"mbps":132.81455427444436,"efficiency_pct":100.0},{"threads":2,"mbps":261.96244459937634,"efficiency_pct":94.36344075669874},{"threads":4,"mbps":471.179096753442,"efficiency_pct":81.99049683695324},{"threads":8,"mbps":797.7341483582311,"efficiency_pct":72.017413348862}],"kitoken":[{"threads":1,"mbps":22.337463925684148,"efficiency_pct":100.0},{"threads":2,"mbps":43.22396239093588,"efficiency_pct":96.27157614898482},{"threads":4,"mbps":81.87895247043213,"efficiency_pct":92.0275161211714},{"threads":8,"mbps":153.94021736341585,"efficiency_pct":82.16778650056207}],"fastokens":[{"threads":1,"mbps":51.52774169537919,"efficiency_pct":100.0},{"threads":2,"mbps":76.4677289756246,"efficiency_pct":77.24767434306406},{"threads":4,"mbps":110.80170377419608,"efficiency_pct":57.22766256279833},{"threads":8,"mbps":147.1877445980241,"efficiency_pct":35.448003026350534}],"gigatoken":[{"threads":1,"mbps":179.5426564702774,"efficiency_pct":100.0},{"threads":2,"mbps":319.71083487422095,"efficiency_pct":92.42164952892996},{"threads":4,"mbps":562.1584191977728,"efficiency_pct":83.53935778244491},{"threads":8,"mbps":1011.2235618329654,"efficiency_pct":76.45875697303079}],"tiktoken":[{"threads":1,"mbps":24.337991781844785,"efficiency_pct":100.0},{"threads":2,"mbps":45.942719003563994,"efficiency_pct":93.65545120917858},{"threads":4,"mbps":82.09435450964244,"efficiency_pct":88.95104616429103},{"threads":8,"mbps":142.51218412644533,"efficiency_pct":76.63168360431305}],"wordchipper":[{"threads":1,"mbps":38.0664238533662,"efficiency_pct":100.0},{"threads":2,"mbps":76.75206241216492,"efficiency_pct":93.81521975663645},{"threads":4,"mbps":149.4472710037792,"efficiency_pct":86.1957914288064},{"threads":8,"mbps":294.19798540489785,"efficiency_pct":73.77300469366004}]},"mt":{"id":"mt","label":"8 threads","panel":"pEight","unit":"MB/s","rows":[{"name":"gigatoken","v":1011,"us":false,"ref":false,"curve":[180,320,562,1011],"eff":76,"range":[63,94],"n":16,"x":23.1},{"name":"tokenizers v1","v":798,"us":true,"ref":false,"curve":[133,262,471,798],"eff":72,"range":[62,89],"n":16,"x":18.3},{"name":"wordchipper","v":294,"us":false,"ref":false,"curve":[38,77,149,294],"eff":74,"range":[58,97],"n":13,"x":6.7},{"name":"kitoken","v":154,"us":false,"ref":false,"curve":[22,43,82,154],"eff":82,"range":[69,95],"n":16,"x":3.5},{"name":"fastokens","v":147,"us":false,"ref":false,"curve":[52,76,111,147],"eff":35,"range":[31,88],"n":14,"x":3.4},{"name":"tiktoken","v":143,"us":false,"ref":false,"curve":[24,46,82,143],"eff":77,"range":[55,86],"n":14,"x":3.3},{"name":"tokenizers 0.23","v":44,"us":false,"ref":true,"curve":[8,15,26,44],"eff":73,"range":[63,83],"n":16,"x":1.0}],"threads":[1,2,4,8],"note":""},"eff":{"id":"eff","label":"scaling","panel":"pScaling","unit":"% of linear","rows":[{"name":"kitoken","v":82,"us":false,"ref":false,"curve":[22,43,82,154],"eff":82,"range":[69,95],"n":16,"x":null},{"name":"tiktoken","v":77,"us":false,"ref":false,"curve":[24,46,82,143],"eff":77,"range":[55,86],"n":14,"x":null},{"name":"gigatoken","v":76,"us":false,"ref":false,"curve":[180,320,562,1011],"eff":76,"range":[63,94],"n":16,"x":null},{"name":"wordchipper","v":74,"us":false,"ref":false,"curve":[38,77,149,294],"eff":74,"range":[58,97],"n":13,"x":null},{"name":"tokenizers 0.23","v":73,"us":false,"ref":true,"curve":[8,15,26,44],"eff":73,"range":[63,83],"n":16,"x":null},{"name":"tokenizers v1","v":72,"us":true,"ref":false,"curve":[133,262,471,798],"eff":72,"range":[62,89],"n":16,"x":null},{"name":"fastokens","v":35,"us":false,"ref":false,"curve":[52,76,111,147],"eff":35,"range":[31,88],"n":14,"x":null}],"max":100,"dec":0,"note":""}}},"scaling_mode_default":"native-threads","mt_all":{"hf-tokenizers":[{"threads":1,"mbps":7.5228663600034,"efficiency_pct":100.0},{"threads":2,"mbps":14.521385608675022,"efficiency_pct":96.7938955950852},{"threads":4,"mbps":26.020517285900514,"efficiency_pct":88.19712617803701},{"threads":8,"mbps":44.0855326530704,"efficiency_pct":77.22426538734213}],"pipeline":[{"threads":1,"mbps":130.19232277600958,"efficiency_pct":100.0},{"threads":2,"mbps":245.10005801837784,"efficiency_pct":95.67243108746871},{"threads":4,"mbps":450.4032565484989,"efficiency_pct":86.24868422308896},{"threads":8,"mbps":832.7926976264296,"efficiency_pct":76.08074670702482}],"fastokens":[{"threads":1,"mbps":153.13629691139056,"efficiency_pct":100.0},{"threads":2,"mbps":100.18596059365922,"efficiency_pct":32.46921028590421},{"threads":4,"mbps":139.41812998827487,"efficiency_pct":22.831554872149084},{"threads":8,"mbps":165.50227756173405,"efficiency_pct":12.974924493501378}],"gigatoken":[{"threads":1,"mbps":176.58144684171077,"efficiency_pct":100.0},{"threads":2,"mbps":105.61840529054237,"efficiency_pct":64.21142226511743},{"threads":4,"mbps":102.98862250812206,"efficiency_pct":31.49840382010632},{"threads":8,"mbps":101.1271577299603,"efficiency_pct":15.87513924631837}]},"latency":{"bytes":512,"rows":[{"name":"deepseek-v4","new_p50":2.292,"new_p99":5.2909999999999995,"rel_p50":87.333,"rel_p99":145.917,"samples":1000,"bytes":512},{"name":"qwen2","new_p50":2.584,"new_p99":5.625,"rel_p50":87.0,"rel_p99":133.5,"samples":1000,"bytes":512},{"name":"minimax","new_p50":2.6670000000000003,"new_p99":5.959,"rel_p50":87.5,"rel_p99":135.875,"samples":1000,"bytes":512},{"name":"gpt2","new_p50":2.458,"new_p99":6.4590000000000005,"rel_p50":77.666,"rel_p99":111.792,"samples":1000,"bytes":512},{"name":"gpt-oss","new_p50":3.375,"new_p99":8.667,"rel_p50":63.87499999999999,"rel_p99":97.958,"samples":1000,"bytes":512},{"name":"glm-5.2","new_p50":2.666,"new_p99":7.625,"rel_p50":60.417,"rel_p99":82.209,"samples":1000,"bytes":512},{"name":"llama-3","new_p50":3.125,"new_p99":10.291,"rel_p50":64.916,"rel_p99":94.66699999999999,"samples":1000,"bytes":512},{"name":"nemotron-3","new_p50":4.0,"new_p99":11.875,"rel_p50":64.5,"rel_p99":108.25,"samples":1000,"bytes":512}]},"key_wins":{"languages":{"unit":"× throughput vs tokenizers 0.23","rows":[{"name":"Amharic","v":14.59,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Hindi","v":12.52,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Bengali","v":11.9,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Arabic","v":11.28,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Korean","v":9.34,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Japanese","v":7.11,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"},{"name":"Chinese","v":6.77,"x":null,"us":true,"ref":false,"n":10,"count_unit":"models"}]},"latency":{"unit":"median p99 for 512-byte English documents","rows":[{"name":"tokenizers v1","v":7.04,"us":true},{"name":"tokenizers 0.23","v":110.02,"ref":true}]},"architectures":{"unit":"× throughput vs tokenizers 0.23","rows":[{"name":"BPE","v":18.02,"n":176,"models":["deepseek-v4","glm-5.2","gpt-oss","gpt2","llama-3","minimax","nemotron-3","qwen2"],"us":true,"ref":false},{"name":"WordPiece","v":6.73,"n":22,"models":["bert-base-uncased"],"us":true,"ref":false},{"name":"Unigram","v":3.33,"n":22,"models":["t5-base"],"us":true,"ref":false}]}},"decode":{"rows":[{"name":"tokie","v":444.2,"x":7.6,"us":false,"ref":false},{"name":"tokenizers v1","v":373.1,"x":6.9,"us":true,"ref":false},{"name":"tiktoken","v":281.2,"x":5.1,"us":false,"ref":false},{"name":"fastokens","v":97.6,"x":1.8,"us":false,"ref":false},{"name":"tokenizers 0.23","v":55.6,"x":1.0,"us":false,"ref":true}],"model_rows":[{"name":"gpt2","new":335.0,"ref":35.6,"x":8.8,"n":22},{"name":"minimax","new":390.5,"ref":52.9,"x":7.4,"n":22},{"name":"glm-5.2","new":389.0,"ref":55.7,"x":7.2,"n":22},{"name":"llama-3","new":375.3,"ref":54.9,"x":7.2,"n":22},{"name":"qwen2","new":395.8,"ref":56.6,"x":7.2,"n":22},{"name":"gpt-oss","new":394.7,"ref":59.7,"x":6.4,"n":22},{"name":"nemotron-3","new":339.3,"ref":56.2,"x":5.7,"n":22},{"name":"deepseek-v4","new":315.9,"ref":55.4,"x":5.4,"n":22}],"reports":1,"cells":130,"models":["glm-5.2","gpt-oss","llama-3","minimax","nemotron-3","qwen2"],"engines":["fastokens","hf-tokenizers","pipeline","tiktoken","tokie"],"versions":{"hf-tokenizers":"0.23.1","pipeline":"tk-encode 1.0.0-rc.0 (tokenizers-rc0 @ 199d9a13)","fastokens":"0.3.1","tokie":"0.1.4","tiktoken":"0.12.0"},"source":"m4"},"memory_modes":{"single":{"label":"1 worker","threads":1,"parallelism":"independent-instances","rows":[{"name":"tokenizers v1","loaded":16.9,"working":19.1,"status":null,"explanation":null,"us":true,"ref":false},{"name":"tiktoken","loaded":31.6,"working":32.0,"status":null,"explanation":null,"us":false,"ref":false},{"name":"rust-gems-bpe","loaded":33.7,"working":34.2,"status":null,"explanation":null,"us":false,"ref":false},{"name":"wordchipper","loaded":43.7,"working":43.7,"status":null,"explanation":null,"us":false,"ref":false},{"name":"tokenizers 0.23","loaded":47.8,"working":51.3,"status":null,"explanation":null,"us":false,"ref":true},{"name":"kitoken","loaded":51.3,"working":51.7,"status":null,"explanation":null,"us":false,"ref":false},{"name":"fastokens","loaded":86.2,"working":89.2,"status":null,"explanation":null,"us":false,"ref":false},{"name":"gigatoken","loaded":87.7,"working":107.1,"status":null,"explanation":null,"us":false,"ref":false},{"name":"tokie","loaded":null,"working":null,"status":"engine has a native batch pool that cannot be pinned to one thread","explanation":null,"us":false,"ref":false}]},"native-threads":{"label":"8 native threads","threads":8,"parallelism":"native-threads","rows":[{"name":"tokenizers v1","loaded":16.9,"working":34.6,"status":null,"explanation":null,"us":true,"ref":false},{"name":"tokenizers 0.23","loaded":47.9,"working":52.8,"status":null,"explanation":null,"us":false,"ref":true},{"name":"fastokens","loaded":86.3,"working":90.5,"status":null,"explanation":null,"us":false,"ref":false},{"name":"gigatoken","loaded":87.8,"working":115.2,"status":null,"explanation":null,"us":false,"ref":false},{"name":"kitoken","loaded":null,"working":null,"status":"engine cannot run with 8 native threads","explanation":null,"us":false,"ref":false},{"name":"tokie","loaded":null,"working":null,"status":"engine cannot run with 8 native threads","explanation":null,"us":false,"ref":false},{"name":"tiktoken","loaded":null,"working":null,"status":"engine cannot run with 8 native threads","explanation":null,"us":false,"ref":false},{"name":"rust-gems-bpe","loaded":null,"working":null,"status":"engine cannot run with 8 native threads","explanation":null,"us":false,"ref":false},{"name":"wordchipper","loaded":null,"working":null,"status":"engine cannot run with 8 native threads","explanation":null,"us":false,"ref":false}]},"independent-instances":{"label":"8 independent instances","threads":8,"parallelism":"independent-instances","rows":[{"name":"tokenizers v1","loaded":135.9,"working":160.4,"status":null,"explanation":null,"us":true,"ref":false},{"name":"tiktoken","loaded":252.3,"working":255.6,"status":null,"explanation":null,"us":false,"ref":false},{"name":"wordchipper","loaded":359.5,"working":359.5,"status":null,"explanation":null,"us":false,"ref":false},{"name":"tokenizers 0.23","loaded":389.8,"working":394.5,"status":null,"explanation":null,"us":false,"ref":true},{"name":"kitoken","loaded":392.6,"working":396.0,"status":null,"explanation":null,"us":false,"ref":false},{"name":"gigatoken","loaded":706.5,"working":707.3,"status":null,"explanation":null,"us":false,"ref":false},{"name":"fastokens","loaded":735.8,"working":739.8,"status":null,"explanation":null,"us":false,"ref":false},{"name":"tokie","loaded":null,"working":null,"status":"engine has a native batch pool that cannot be pinned to one thread","explanation":null,"us":false,"ref":false},{"name":"rust-gems-bpe","loaded":null,"working":null,"status":"shared process-global tokenizer","explanation":"All handles share one process-global tokenizer, so this does not measure eight independent tokenizer allocations.","us":false,"ref":false}]}},"memory_default":"single","crate_sizes":{"unit":"gzipped executable bytes","profile":"minsize, stripped, gzip -9","platform":"Apple M4 Max, aarch64-apple-darwin","rustc":"rustc 1.97.1 (8bab26f4f 2026-07-14)","tokenizers_revision":"484be1d3351077fc5c0b81e876c94814c08ef3eb","initial_revision":"0e9bb9ca3bfebcc3892173a788d5fcda9e7bf2cd","baseline":{"name":"tokenizers before the split","bytes":665263},"required":"tk-encode","options":["serialize","convert","train"],"configs":{"encode":305989,"train":424801,"convert":333417,"convert+train":452621,"serialize":338003,"serialize+train":456838,"serialize+convert":362199,"serialize+convert+train":481005},"feature_required":"bpe","feature_options":["unigram","wordpiece","wordlevel","normalizers","unicode-scripts","parallelism"],"feature_profiles":{"minsize":{"bpe":305989,"normalizers":399026,"normalizers+parallelism":404239,"normalizers+unicode-scripts+parallelism":417243,"unigram+normalizers+unicode-scripts+parallelism":431505,"unigram+wordlevel+normalizers+unicode-scripts+parallelism":433401,"unigram+wordpiece+wordlevel+normalizers+unicode-scripts+parallelism":435335,"unigram+wordpiece+normalizers+unicode-scripts+parallelism":434424,"wordlevel+normalizers+unicode-scripts+parallelism":418704,"wordpiece+wordlevel+normalizers+unicode-scripts+parallelism":420405,"wordpiece+normalizers+unicode-scripts+parallelism":419468,"unigram+normalizers+parallelism":418766,"unigram+wordlevel+normalizers+parallelism":419848,"unigram+wordpiece+wordlevel+normalizers+parallelism":421389,"unigram+wordpiece+normalizers+parallelism":420678,"wordlevel+normalizers+parallelism":405989,"wordpiece+wordlevel+normalizers+parallelism":407779,"wordpiece+normalizers+parallelism":406834,"normalizers+unicode-scripts":414056,"unigram+normalizers+unicode-scripts":426450,"unigram+wordlevel+normalizers+unicode-scripts":429732,"unigram+wordpiece+wordlevel+normalizers+unicode-scripts":431366,"unigram+wordpiece+normalizers+unicode-scripts":430415,"wordlevel+normalizers+unicode-scripts":415463,"wordpiece+wordlevel+normalizers+unicode-scripts":416753,"wordpiece+normalizers+unicode-scripts":415099,"unigram+normalizers":414907,"unigram+wordlevel+normalizers":416649,"unigram+wordpiece+wordlevel+normalizers":417255,"unigram+wordpiece+normalizers":417268,"wordlevel+normalizers":401201,"wordpiece+wordlevel+normalizers":402907,"wordpiece+normalizers":402024,"parallelism":309738,"unicode-scripts+parallelism":324291,"unigram+unicode-scripts+parallelism":338991,"unigram+wordlevel+unicode-scripts+parallelism":340327,"unigram+wordpiece+wordlevel+unicode-scripts+parallelism":342200,"unigram+wordpiece+unicode-scripts+parallelism":341408,"wordlevel+unicode-scripts+parallelism":326132,"wordpiece+wordlevel+unicode-scripts+parallelism":327627,"wordpiece+unicode-scripts+parallelism":326869,"unigram+parallelism":325499,"unigram+wordlevel+parallelism":325818,"unigram+wordpiece+wordlevel+parallelism":327459,"unigram+wordpiece+parallelism":326659,"wordlevel+parallelism":311622,"wordpiece+wordlevel+parallelism":313337,"wordpiece+parallelism":312444,"unicode-scripts":319351,"unigram+unicode-scripts":333534,"unigram+wordlevel+unicode-scripts":334903,"unigram+wordpiece+wordlevel+unicode-scripts":336888,"unigram+wordpiece+unicode-scripts":335695,"wordlevel+unicode-scripts":321033,"wordpiece+wordlevel+unicode-scripts":322603,"wordpiece+unicode-scripts":321756,"unigram":320535,"unigram+wordlevel":321995,"unigram+wordpiece+wordlevel":323636,"unigram+wordpiece":322474,"wordlevel":307433,"wordpiece+wordlevel":308886,"wordpiece":308115},"slimest":{"bpe":164189,"normalizers":256852,"normalizers+parallelism":260034,"normalizers+unicode-scripts+parallelism":273030,"unigram+normalizers+unicode-scripts+parallelism":285066,"unigram+wordlevel+normalizers+unicode-scripts+parallelism":286382,"unigram+wordpiece+wordlevel+normalizers+unicode-scripts+parallelism":288043,"unigram+wordpiece+normalizers+unicode-scripts+parallelism":287351,"wordlevel+normalizers+unicode-scripts+parallelism":275286,"wordpiece+wordlevel+normalizers+unicode-scripts+parallelism":276833,"wordpiece+normalizers+unicode-scripts+parallelism":276144,"unigram+normalizers+parallelism":272541,"unigram+wordlevel+normalizers+parallelism":273527,"unigram+wordpiece+wordlevel+normalizers+parallelism":275422,"unigram+wordpiece+normalizers+parallelism":274580,"wordlevel+normalizers+parallelism":261979,"wordpiece+wordlevel+normalizers+parallelism":263977,"wordpiece+normalizers+parallelism":263066,"normalizers+unicode-scripts":270163,"unigram+normalizers+unicode-scripts":281967,"unigram+wordlevel+normalizers+unicode-scripts":282168,"unigram+wordpiece+wordlevel+normalizers+unicode-scripts":285279,"unigram+wordpiece+normalizers+unicode-scripts":284251,"wordlevel+normalizers+unicode-scripts":271673,"wordpiece+wordlevel+normalizers+unicode-scripts":273438,"wordpiece+normalizers+unicode-scripts":272561,"unigram+normalizers":267958,"unigram+wordlevel+normalizers":270588,"unigram+wordpiece+wordlevel+normalizers":272515,"unigram+wordpiece+normalizers":270694,"wordlevel+normalizers":258290,"wordpiece+wordlevel+normalizers":260215,"wordpiece+normalizers":259390,"parallelism":167971,"unicode-scripts+parallelism":182186,"unigram+unicode-scripts+parallelism":194301,"unigram+wordlevel+unicode-scripts+parallelism":195495,"unigram+wordpiece+wordlevel+unicode-scripts+parallelism":197600,"unigram+wordpiece+unicode-scripts+parallelism":196725,"wordlevel+unicode-scripts+parallelism":184050,"wordpiece+wordlevel+unicode-scripts+parallelism":185624,"wordpiece+unicode-scripts+parallelism":184982,"unigram+parallelism":181000,"unigram+wordlevel+parallelism":181135,"unigram+wordpiece+wordlevel+parallelism":183105,"unigram+wordpiece+parallelism":182311,"wordlevel+parallelism":169861,"wordpiece+wordlevel+parallelism":171557,"wordpiece+parallelism":170841,"unicode-scripts":177253,"unigram+unicode-scripts":189216,"unigram+wordlevel+unicode-scripts":190692,"unigram+wordpiece+wordlevel+unicode-scripts":192771,"unigram+wordpiece+unicode-scripts":191823,"wordlevel+unicode-scripts":178735,"wordpiece+wordlevel+unicode-scripts":181076,"wordpiece+unicode-scripts":180238,"unigram":176277,"unigram+wordlevel":177571,"unigram+wordpiece+wordlevel":179387,"unigram+wordpiece":178595,"wordlevel":165783,"wordpiece+wordlevel":167706,"wordpiece":166985}},"matrix_tool":"cargo-matrix 0.4.5","absolute_minimum":{"bytes":164189,"profile":"slimest, rebuilt std with panic_immediate_abort","rustc":"rustc 1.99.0-nightly (1ed2df61a 2026-08-04)"}},"cache_gain":{"added_special_dense":1.69,"agentic_swe":1.38,"cmn_Hani":0.98,"code_mixed":1.3,"eng_Latn":0.93,"math_latex":1.11,"prefix_sharing":2.03},"cache_meta":{"reports":1,"cells":48,"models":8,"versions":{"pipeline":"tk-encode 1.0.0-rc.0 (tokenizers-rc0 @ 5c3727a9)","pipeline-no-cache":"tk-encode 1.0.0-rc.0 (tokenizers-rc0 @ 5c3727a9)"}},"cache_previews":{"added_special_dense":{"segments":[{"kind":"sample","text":"fast <|xs2|> chunk <|xs1|> <|xs4|> <|xs3|> <|xs4|> modality language <|xs0|> reads <|xs2|> and and \n <|xs1|> <|xs0|> <|xs1|> <|xs4|> <|xs4|> <|xs1|> back and reads corpus <|xs3|> <|xs3|> <|xs3|> \n <|xs1|> <|xs3|> <|xs2|> <|xs3|> again <|xs4|> <|xs3|> <|xs2|> the normalize for <|xs4|> <|xs4|> \n <|xs4|> <|xs1|> <|xs2|> we the <|xs1|> text <|xs0|> <|xs1|> <|xs0|> flows merge fast \n <|xs0|> <|xs2|> <|xs0|> <|xs0|> <|xs0|> <|xs1|> the <|xs4|> <|xs3|> <|xs1|> <|xs0|> <|xs4|> <|xs0|> \n <|xs3|> <|xs3|> <|xs4|> <|xs2|> through <|xs1|> <|xs2|> model <|xs1|> <|xs3|> <|xs2|> <|xs2|> <|xs4|> \n model <|xs1|> <|xs2|> <|xs1|> <|xs3|> <|xs4|> runs decode <|xs0|> <|xs4|> <|xs4|> <|xs2|> <|xs2|> \n <|xs0|> <|xs3|> <|xs0|> <|xs0|> <|xs0|> <|xs1|> <|xs0|> <|xs3|> <|xs4|> <|xs1|> <|xs1|> <|xs3|> <|xs0|> \n <|xs0|> and <|xs0|> <|xs3|> <|xs4|> input <|xs2|> <|xs2|> <|xs3|> <|xs2|> <|xs0|> fast <|xs3|> \n <|xs4|> <"}]},"agentic_swe":{"segments":[{"kind":"sample","text":"[system]\nYou are a helpful assistant that can interact with a computer to solve tasks.\n\n[user]\n<uploaded_files>\n/testbed\n<\/uploaded_files>\nI've uploaded a python code repository in the directory /testbed. Consider the following PR description:\n\n<pr_description>\n# MoneyWidget decompress method breaks form validation with disabled fields\n\nI've discovered an issue with the `MoneyWidget` class in the forms/widgets.py file. When a form field using this widget is disabled, validation fails unexpectedly.\n\n## Steps to reproduce\n\n1. Create a model with a MoneyField:\n```python\nclass ModelWithVanillaMoneyField(models.Model):\n money = MoneyField(max_digits=10, decimal_places=2, default_currency='USD')\n```\n\n2. Create a form with a disabled money field:\n```python\nclass DisabledFieldForm(forms.ModelForm):\n class Meta:\n model = ModelWithVanillaMoneyField\n fields = ('money',)\n "}]},"cmn_Hani":{"segments":[{"kind":"sample","text":"強力建議大家未來盡量避免英航阿~~~\n班機原訂航程\n6/9 台北→香港\n6/9 香港→倫敦 (原訂23:15起飛,4:50am抵達)\n6/10 倫敦→斯德哥爾摩 (原訂7:40am起飛,11:05am抵達)\n就在第二段,英航no.25班機在香港機場兩度離開閘口又兩度返航\n第一次因有旅客身體不適 (好,不怪他,消耗時間也不多,空服員說不影響抵達時間)\n第二次機長廣播說因為飛機某座位傳出不明電器燒焦味! 沒錯!燒焦味!(electrical burning smell)\n眼看時間一直快轉,機艙悶熱,大家都等到頭昏昏了,最後一次看時間已經超過凌晨2點\n但查那麼久,最可怕的是還查不到原因\n機長廣播了幾次進度,最後說,查不到原因,但燒焦味不見了,所以還是起飛吧~~~\n於是,在delay 3小時候,我們終於起飛\n班機飛行中途,抓住了一個空服員問「我們要轉機鐵定搭不上,該怎麼辦哩?」\n空服員臉酷酷,說下飛機後去找ticket desk → 自己處理的意思\n當下覺得他們還挺沒效率的,班機delay那麼久,肯定影響很多旅客阿\n他們不能請其他地勤幫忙處理一下轉機旅客重新預定的問題咩…\n飛機還要在空中飛10幾小時,非得要旅客等下飛機再去處理?\n心懷不滿但我這輩子還沒有被delay到搭不上轉機飛機\n想說搞不好SOP就是這樣子\n沒想到,在降落前半小時,機長傳來令人振奮的廣播\n他說他們知道大家都很擔心轉機,他們已經幫各旅客 “Re-booking”轉機班機\n接下來開始一個一個廣播旅客姓名、班機名稱和時間…\n廣播長達10幾分鐘,太多旅客要轉機啦!\n我非常雀躍的拿著筆抄下號碼,噢!改11:30點起飛,那也還ok\n機長還貼心提醒大家要去ticket desk,collect boarding pass\nGreat!!\n再次沒想到\n下飛機找到ticket desk時,卻全..然..不...是..那..回..事!!!\n英航人員查電腦紀錄說,沒有re-booking阿!沒有re-booking阿!沒有re-booking阿!\n(蝦密!)\n更慘的是,11:30班機滿了,13:30班機也滿了,要等到5點多那班\n心"}]},"code_mixed":{"segments":[{"kind":"sample","text":"// django-f3f9601cff03b38942e70ce8042bdfdec6002449/django/__init__.py\nfrom django.utils.version import get_version\n\nVERSION = (6, 2, 0, \"alpha\", 0)\n\n__version__ = get_version(VERSION)\n\n\ndef setup(set_prefix=True):\n \"\"\"\n Configure the settings (this happens as a side effect of accessing the\n first setting), configure logging and populate the app registry.\n Set the thread-local urlresolvers script prefix if `set_prefix` is True.\n \"\"\"\n from django.apps import apps\n from django.conf import settings\n from django.urls import set_script_prefix\n from django.utils.log import configure_logging\n\n configure_logging(settings.LOGGING_CONFIG, settings.LOGGING)\n if set_prefix:\n set_script_prefix(\n \"/\" if settings.FORCE_SCRIPT_NAME is None else settings.FORCE_SCRIPT_NAME\n )\n apps.populate(settings.INSTALLED_APPS)\n\n// django-f3f9601cff03b38942e7"}]},"eng_Latn":{"segments":[{"kind":"sample","text":"|Viewing Single Post From: Spoilers for the Week of February 11th|\n|Lil||Feb 1 2013, 09:58 AM|\nDon't care about Chloe/Taniel/Jen-Jen. Don't care about Sami, really, but hoping that we get some good \"SAMANTHA GENE!!\" Marlena Death-Stares out of it. And \"newfound\" feelings. Please. If only.\nSTEFANO!! STEFANO, STEFANO, STEFANO!!!! :cheer:\n|Spoilers for the Week of February 11th · DAYS: News, Spoilers & Discussion|\n\n*sigh* Fundamentalist community, let me pass on some advice to you I learned from the atheistic community:\nIf you have set yourself on fire, do not run.\nOkay? Okay?? Please?\nLook, D, you had two months to say to Harvard in private emails, \"Im sorry, I shouldnt have been using that animation in my paid presentations. I wont use it again. I really do like 'Inner Life', though, and would love to use it in classroom presentations, from the BioVisions site, if that is acceptable.\"\nI s"}]},"math_latex":{"segments":[{"kind":"sample","text":"Bayes and his Theorem\n\nMy earlier post on Bayesian probability seems to have generated quite a lot of readers, so this lunchtime I thought I’d add a little bit of background. The previous discussion started from the result\n\n$P(B|AC) = K^{-1}P(B|C)P(A|BC) = K^{-1} P(AB|C)$\n\nwhere\n\n$K=P(A|C).$\n\nAlthough this is called Bayes’ theorem, the general form of it as stated here was actually first written down, not by Bayes but by Laplace. What Bayes’ did was derive the special case of this formula for “inverting” the binomial distribution. This distribution gives the probability of x successes in n independent “trials” each having the same probability of success, p; each “trial” has only two possible outcomes (“success” or “failure”). Trials like this are usually called Bernoulli trials, after Daniel Bernoulli. If we ask the question “what is the probability of exactly x successes from the possib"}]},"prefix_sharing":{"segments":[{"kind":"prefix","text":"[system]\nYou are a helpful assistant that can interact with a computer to solve tasks.\n\n[user]\n<uploaded_files>\n/testbed\n<\/uploaded_files>\nI've uploaded a python code repository in the directory /testbed. Consider the following PR description:\n\n<pr_description>\n# MoneyWidget decompress method breaks form validation with disabled fields\n\nI've discovered an issue with the `MoneyWidget` class in the forms/widgets.py file. When a form field using this widget is disabled, validation fails unexpectedly.\n\n## Steps to reproduce\n\n1. Create a model with a MoneyField:\n```python\nclass ModelWithVanillaMoneyField(models.Model):\n money = MoneyField(max_digits=10, decimal_places=2, default_currency='USD')\n```\n\n2. Create a form with a disabled money field:\n```python\nclass DisabledFieldForm(forms.ModelForm):\n class Meta:\n model = ModelWithVanillaMoneyField\n fields = ('money',)\n "},{"kind":"suffix","text":"ons, but you can't do following:\r\n 45\t - Add or subtract money with not-money\r\n 46\t - Any exponentiation\r\n 47\t - Any operations with money in different currencies\r\n 48\t - Multiplication, division, modulo with money instances on both sides of expression\r\n 49\t \"\"\"\r\n 50\t connector = expr.connector\n\n[assistant]\nLet's continue looking at the `MoneyField` class:\n[{\"index\": 1, \"function\": {\"arguments\": \"{\\\"command\\\": \\\"view\\\", \\\"path\\\": \\\"/testbed/djmoney/models/fields.py\\\", \\\"view_range\\\": [100, 150]}\", \"name\": \"str_replace_editor\"}, \"id\": \"toolu_01SC44ouZ"}]}},"demo":[{"m":"deepseek-v4","x":"The tokenizer is no longer the bottleneck.","n":"The tokenizer is no longer the bottleneck.","nchanged":false,"ntype":"Sequence","b":["The","Ġtokenizer","Ġis","Ġno","Ġlonger","Ġthe","Ġbottleneck","."],"t":["The","Ġtoken","izer","Ġis","Ġno","Ġlonger","Ġthe","Ġbottleneck","."],"ib":["671","17840","9160","344","1119","5827","270","111127","16"],"i":["671","17840","9160","344","1119","5827","270","111127","16"],"added":[],"addpre":[],"addpost":[],"nkind":null,"fast":true,"marker":"Ġ"},{"m":"llama-3","x":"The tokenizer is no longer the bottleneck.","n":"The tokenizer is no longer the bottleneck.","nchanged":false,"ntype":null,"b":["The","Ġtokenizer","Ġis","Ġno","Ġlonger","Ġthe","Ġbottleneck","."],"t":["The","Ġtokenizer","Ġis","Ġno","Ġlonger","Ġthe","Ġbottleneck","."],"ib":["791","47058","374","912","5129","279","88938","13"],"i":["128000","791","47058","374","912","5129","279","88938","13"],"added":["<|begin_of_text|>"],"addpre":["<|begin_of_text|>"],"addpost":[],"nkind":null,"fast":true,"marker":"Ġ"},{"m":"bert","x":"The tokenizer is no longer the bottleneck.","n":"the tokenizer is no longer the bottleneck.","nchanged":true,"ntype":"BertNormalizer","b":["the","tokenizer","is","no","longer","the","bottleneck","."],"t":["the","token","##izer","is","no","longer","the","bottle","##neck","."],"ib":["1996","19204","17629","2003","2053","2936","1996","5835","18278","1012"],"i":["101","1996","19204","17629","2003","2053","2936","1996","5835","18278","1012","102"],"added":["[CLS]","[SEP]"],"addpre":["[CLS]"],"addpost":["[SEP]"],"nkind":"lower","fast":false,"marker":"##"}],"titles":{"results":{"x":"results","p":["results"],"b":["results"],"t":["results"],"i":["47115"]},"what v1 is":{"x":"what v1 is","p":["what"," v","1"," is"],"b":["what","Ġv","1","Ġis"],"t":["what","Ġv","1","Ġis"],"i":["9602","374","19","344"]},"the split: bitstreams instead of a regex":{"x":"the split: bitstreams instead of a regex","p":["the"," split",":"," bitstreams"," instead"," of"," a"," regex"],"b":["the","Ġsplit",":","Ġbitstreams","Ġinstead","Ġof","Ġa","Ġregex"],"t":["the","Ġsplit",":","Ġbit","stream","s","Ġinstead","Ġof","Ġa","Ġregex"],"i":["1805","14241","28","4669","10628","85","6240","294","260","65327"]},"the word cache":{"x":"the word cache","p":["the"," word"," cache"],"b":["the","Ġword","Ġcache"],"t":["the","Ġword","Ġcache"],"i":["1805","2004","23809"]},"the merge loop":{"x":"the merge loop","p":["the"," merge"," loop"],"b":["the","Ġmerge","Ġloop"],"t":["the","Ġmerge","Ġloop"],"i":["1805","29446","12175"]},"method":{"x":"method","p":["method"],"b":["method"],"t":["method"],"i":["23735"]},"what this adds up to":{"x":"what this adds up to","p":["what"," this"," adds"," up"," to"],"b":["what","Ġthis","Ġadds","Ġup","Ġto"],"t":["what","Ġthis","Ġadds","Ġup","Ġto"],"i":["9602","566","16803","890","304"]},"getting it":{"x":"getting it","p":["getting"," it"],"b":["getting","Ġit"],"t":["getting","Ġit"],"i":["85703","436"]},"progress towards v1":{"x":"progress towards v1","p":["progress"," towards"," v","1"],"b":["progress","Ġtowards","Ġv","1"],"t":["progress","Ġtowards","Ġv","1"],"i":["72618","6104","374","19"]},"release candidate: implemented":{"x":"release candidate: implemented","p":["release"," candidate",":"," implemented"],"b":["release","Ġcandidate",":","Ġimplemented"],"t":["release","Ġcandidate",":","Ġimplemented"],"i":["90660","14626","28","14315"]},"1.0.0":{"x":"1.0.0","p":["1",".","0",".","0"],"b":["1",".","0",".","0"],"t":["1",".","0",".","0"],"i":["19","16","18","16","18"]},"after 1.0.0":{"x":"after 1.0.0","p":["after"," ","1",".","0",".","0"],"b":["after","Ġ","1",".","0",".","0"],"t":["after","Ġ","1",".","0",".","0"],"i":["15479","223","19","16","18","16","18"]}}}</script> | |
| <script> | |
| ; | |
| const D = JSON.parse(document.getElementById("data").textContent); | |
| const $ = s => document.querySelector(s); | |
| const esc = s => String(s).replace(/[&<>]/g, c => ({"&":"&","<":"<",">":">"}[c])); | |
| const attr = s => esc(s).replace(/"/g, """); | |
| const pad = (s, n) => (s + " ".repeat(Math.max(0, n - s.length))); | |
| const lpad = (s, n) => (" ".repeat(Math.max(0, n - String(s).length)) + s); | |
| const reduced = matchMedia("(prefers-reduced-motion: reduce)").matches; | |
| /* ---- animations ------------------------------------------------------- */ | |
| const running = {}; | |
| // step(t, dir): t is the running frame count, dir is which way this call moves. | |
| // dir is 0 on the first paint, 1 for each interval tick, and -1 or 1 when a | |
| // reader uses the step buttons. An animation that just counts can ignore dir; an | |
| // animation a reader can walk backwards through needs it, because t alone cannot | |
| // say whether the last move was forward or back. | |
| function toggler(id, step, fps){ | |
| let t = 0, on = !reduced; | |
| running[id] = on; | |
| const btn = document.querySelector(`[data-toggle="${id}"]`); | |
| const label = () => { if (btn) btn.textContent = running[id] ? "pause" : "play"; }; | |
| if (btn) btn.onclick = () => { running[id] = !running[id]; label(); }; | |
| if (btn && reduced) btn.textContent = "play"; | |
| // Stepping implies wanting to look at a frame, so it pauses first. | |
| document.querySelectorAll(`[data-step="${id}"]`).forEach(b => { | |
| const dir = Number(b.dataset.dir) || 1; | |
| b.onclick = () => { running[id] = false; label(); t += dir; step(t, dir); }; | |
| }); | |
| step(0, 0); | |
| setInterval(() => { if (running[id]) step(++t, 1); }, 1000 / fps); | |
| } | |
| /* ---- decorative rules ------------------------------------------------- */ | |
| function fillRules(){ | |
| document.querySelectorAll(".rule").forEach(e => e.textContent = "─".repeat(96)); | |
| document.querySelectorAll("h2 .fill").forEach(e => e.textContent = "─".repeat(90)); | |
| } | |
| fillRules(); | |
| /* ---- ascii bar chart -------------------------------------------------- */ | |
| const BLK = "█", MED = "▓", LGT = "░"; | |
| // Bars are drawn in characters, so their width has to be derived from the real | |
| // character advance of the rendered font rather than assumed. | |
| let CHW = 8; | |
| function measureCh(){ | |
| const t = document.createElement("span"); | |
| t.style.cssText = "position:absolute;visibility:hidden;white-space:pre;font:inherit"; | |
| t.textContent = "0".repeat(100); | |
| document.body.appendChild(t); | |
| const w = t.getBoundingClientRect().width / 100; | |
| t.remove(); | |
| if (w > 0) CHW = w; | |
| } | |
| // columns left over for the bar once label, value and suffix have taken theirs | |
| function fitW(el, used){ | |
| const px = (el && el.clientWidth) || 640; | |
| return Math.max(12, Math.min(180, Math.floor(px / CHW) - used - 2)); | |
| } | |
| function bars(rows, opts){ | |
| opts = opts || {}; | |
| DEC = opts.dec !== undefined ? opts.dec : decFor(rows); | |
| const labw = Math.max(...rows.map(r => r.name.length)) + 1; | |
| const valw = Math.max(...rows.map(r => fmt(r.v).length)); | |
| const count = r => r.n ? `${r.n} ${r.count_unit || "corpora"}` : ""; | |
| const tailw = Math.max(...rows.map(r => { | |
| const value = r.x ? 3 + r.x.toFixed(1).length : count(r) ? 2 + count(r).length : 0; | |
| return value + (r.action ? r.action.length + 5 : 0); | |
| })); | |
| let W = fitW(opts.el, labw + valw + tailw + 1); | |
| if (opts.pivot) W -= W % 2; // a diverging bar needs an exact centre column | |
| const max = opts.max || Math.max(...rows.map(r => r.v)); | |
| const pivot = opts.pivot; | |
| // the unit sits over the bar column, not at the left margin where it reads as prose | |
| let out = opts.unit | |
| ? `<span class="row unit">${esc(opts.unit)}</span>` | |
| : ""; | |
| for (const r of rows){ | |
| const cls = r.us ? " us" : (r.ref ? " ref" : ""); | |
| const inspect = r.inspect | |
| ? ` class="row inspectable" role="button" tabindex="0" aria-expanded="false" aria-controls="${esc(r.inspect)}" data-inspect="${esc(r.inspect)}"${r.previewKey ? ` data-preview-key="${esc(r.previewKey)}"` : ""}` | |
| : ` class="row"`; | |
| let bar; | |
| if (pivot){ | |
| // diverging around 1.0: centre column, right = v1 ahead | |
| const half = Math.floor(W / 2); | |
| const span = Math.max(...rows.map(x => Math.abs(x.v - pivot))) || 1; | |
| const n = Math.round(Math.abs(r.v - pivot) / span * half); | |
| bar = r.v >= pivot | |
| ? " ".repeat(half) + "│" + (r.us ? BLK : MED).repeat(n) | |
| + " ".repeat(Math.max(0, half - n)) | |
| : " ".repeat(Math.max(0, half - n)) + MED.repeat(n) + "│" + " ".repeat(half); | |
| } else { | |
| const n = Math.min(W, Math.max(1, Math.round(r.v / max * W))); | |
| const g = r.us ? BLK : (r.ref ? LGT : MED); | |
| bar = g.repeat(n) + " ".repeat(W - n); | |
| } | |
| out += `<span${inspect}><span class="lab${cls}">${esc(pad(r.name, labw))}</span>` | |
| + `<span class="bar${r.us ? " us" : ""}">${bar}</span> ` | |
| + `<span class="val${r.us ? " us" : ""}">${lpad(fmt(r.v), valw)}</span>` | |
| + (r.x ? `<span class="x"> ×${r.x.toFixed(1)}</span>` : (count(r) ? `<span class="x"> ${esc(count(r))}</span>` : "")) | |
| + (r.action ? `<span class="inspect-mark"> [ ${esc(r.action)} ]</span>` : "") | |
| + `</span>\n`; | |
| } | |
| DEC = null; | |
| return out; | |
| } | |
| // decimals are chosen per column, not per value, so the numbers line up | |
| let DEC = null; | |
| const fmt = v => v.toFixed(DEC !== null ? DEC : (Number.isInteger(v) ? 0 : v < 10 ? 2 : 1)); | |
| const decFor = rows => { | |
| const m = Math.max(...rows.map(r => Math.abs(r.v))); | |
| return m >= 1000 ? 0 : m >= 100 ? 1 : 2; | |
| }; | |
| /* ---- tabs ------------------------------------------------------------- */ | |
| let cur = 0; | |
| let selectedBaseline = "hf-tokenizers"; | |
| let selectedScalingMode = D.scaling_mode_default || "native-threads"; | |
| let selectedMemoryMode = D.memory_default || "single"; | |
| let lastAnimatedTab = -1; | |
| $("#baselineSelect").onchange = event => { | |
| selectedBaseline = event.target.value; | |
| renderChart(); | |
| }; | |
| $("#scalingModeSelect").value = selectedScalingMode; | |
| $("#scalingModeSelect").onchange = event => { | |
| selectedScalingMode = event.target.value; | |
| renderChart(); | |
| }; | |
| $("#memoryModeSelect").value = selectedMemoryMode; | |
| $("#memoryModeSelect").onchange = event => { | |
| selectedMemoryMode = event.target.value; | |
| renderMemory(); | |
| }; | |
| // built once; selecting a tab only flips aria-selected, so focus survives a click | |
| // and arrow-key navigation has a stable set of buttons to move between. | |
| function renderTabs(){ | |
| $("#tabs").innerHTML = D.tabs.map((t, i) => | |
| `<button role="tab" data-i="${i}" aria-selected="${i === cur}" tabindex="${i === cur ? 0 : -1}" | |
| >${esc(t.label)}</button>`).join(""); | |
| const btns = [...$("#tabs").querySelectorAll("button")]; | |
| const select = i => { | |
| cur = i; | |
| btns.forEach((b, j) => { b.setAttribute("aria-selected", j === i); b.tabIndex = j === i ? 0 : -1; }); | |
| renderChart(); | |
| }; | |
| btns.forEach((b, i) => { | |
| b.onclick = () => select(i); | |
| b.onkeydown = e => { | |
| const d = e.key === "ArrowRight" ? 1 : e.key === "ArrowLeft" ? -1 : 0; | |
| if (!d) return; | |
| e.preventDefault(); | |
| const n = (i + d + btns.length) % btns.length; | |
| select(n); btns[n].focus(); | |
| }; | |
| }); | |
| } | |
| function renderChart(){ | |
| const base = D.tabs[cur]; | |
| const scaling = base.id === "mt" || base.id === "eff"; | |
| const mode = D.scaling_modes?.[selectedScalingMode]; | |
| const t = scaling && mode ? mode[base.id] : base; | |
| $("#scalingModeControl").hidden = !scaling; | |
| $("#scalingModeHelp").hidden = !scaling; | |
| const want = t.panel || "pBars"; | |
| const panels = [...document.querySelectorAll("#ranking .panel")]; | |
| panels.forEach(p => { p.hidden = p.id !== want; }); | |
| if (lastAnimatedTab !== cur){ | |
| const active = document.getElementById(want); | |
| active.classList.remove("tab-enter"); | |
| void active.offsetWidth; | |
| active.classList.add("tab-enter"); | |
| lastAnimatedTab = cur; | |
| } | |
| if (!t.panel){ | |
| let view = t; | |
| const control = $("#baselineControl"); | |
| if (t.comparisons){ | |
| if (!t.comparisons[selectedBaseline]) selectedBaseline = t.default_baseline; | |
| const select = $("#baselineSelect"); | |
| select.innerHTML = Object.entries(t.comparisons).map(([key, item]) => | |
| `<option value="${esc(key)}">${esc(item.label)}</option>`).join(""); | |
| select.value = selectedBaseline; | |
| control.hidden = false; | |
| view = t.comparisons[selectedBaseline]; | |
| } else { | |
| control.hidden = true; | |
| } | |
| $("#chart").innerHTML = bars(view.rows, | |
| {max: view.max, pivot: t.pivot, unit: view.unit, dec: view.dec, el: $("#chart")}); | |
| $("#chartNote").textContent = view.note || ""; | |
| $("#chartNote").hidden = !view.note; | |
| } else if (t.panel === "pEight"){ | |
| $("#eightChart").innerHTML = bars(t.rows, | |
| {max: t.max, unit: t.unit, dec: t.dec, el: $("#eightChart")}); | |
| $("#eightNote").textContent = t.note || ""; | |
| $("#eightNote").hidden = !t.note; | |
| } else if (t.panel === "pScaling" && typeof renderThreadLines === "function"){ | |
| renderThreadLines(); // now measurable, so it gets the real width | |
| } else if (t.panel === "pDecode" && typeof renderDecode === "function"){ | |
| renderDecode(); // hidden panels have no measurable width | |
| } else if (t.panel === "pMemory" && typeof renderMemory === "function"){ | |
| renderMemory(); | |
| } else if (t.panel === "pCrates" && typeof renderCrates === "function"){ | |
| renderCrates(); | |
| } | |
| } | |
| renderTabs(); renderChart(); | |
| /* ---- hero stats ------------------------------------------------------- */ | |
| // hero numbers are substituted into copy.md at build time, not set here | |
| // p99 tail speed-up, the range across every model measured | |
| const hero = [ | |
| // the span across models, not a median: how much you gain depends on the model | |
| ["k", "×" + (D.meta.hero_lo ?? D.meta.hero_x) + "–" + (D.meta.hero_hi ?? D.meta.hero_x), | |
| "vs 0.23, single thread"], | |
| ["", (D.meta.eff_v1 ?? 94) + "%", "of linear scaling to 8 cores"], | |
| ]; | |
| if (D.meta.lat_lo !== undefined && D.meta.lat_hi !== undefined) | |
| hero.push(["", "\u00d7" + D.meta.lat_lo + "\u2013" + D.meta.lat_hi, | |
| "faster p99 encode latency"]); | |
| if (D.meta.memory_loaded_ratio !== undefined) | |
| hero.push(["", D.meta.memory_loaded_ratio + "\u00d7", "lower loaded heap"]); | |
| if (D.meta.slimest_reduction !== undefined) | |
| hero.push(["", D.meta.slimest_reduction + "\u00d7", "smaller absolute slimest executable"]); | |
| $("#heroStats").innerHTML = hero.map(([k, v, l]) => | |
| `<div class="stat ${k}"><span class="v">${v}</span><span class="l">${l}</span></div>` | |
| ).join(""); | |
| /* ---- three measured wins --------------------------------------------- */ | |
| function renderKeyWins(){ | |
| const K = D.key_wins; | |
| if (!K) return; | |
| const miniBars = (rows, unit, suffix) => { | |
| const max = Math.max(...rows.map(row => row.v)); | |
| return `<span class="unit">${esc(unit)}</span>` + rows.map(row => { | |
| const width = Math.max(2, row.v / max * 100); | |
| return `<div class="mini-row${row.us ? " us" : ""}">` | |
| + `<span class="name">${esc(row.name)}</span>` | |
| + `<span class="track"><span class="fill" style="width:${width.toFixed(1)}%"></span></span>` | |
| + `<span class="value">${row.v.toFixed(1)}${suffix}</span></div>`; | |
| }).join(""); | |
| }; | |
| $("#keyLanguageChart").innerHTML = miniBars(K.languages.rows, K.languages.unit, "×"); | |
| $("#keyLatencyChart").innerHTML = miniBars(K.latency.rows, K.latency.unit, " µs"); | |
| $("#keyArchitectureChart").innerHTML = miniBars( | |
| K.architectures.rows, K.architectures.unit, "×" | |
| ); | |
| } | |
| renderKeyWins(); | |
| /* ---- decode throughput ------------------------------------------------ */ | |
| function renderDecode(){ | |
| if (!D.decode) return; | |
| $("#decodeChart").innerHTML = bars(D.decode.rows, | |
| {unit: "decoded UTF-8 output · MB/s", el: $("#decodeChart")}); | |
| } | |
| renderDecode(); | |
| /* ---- cache effect ---------------------------------------------------- */ | |
| function renderCache(){ | |
| const gain = D.cache_gain || {}; | |
| if (!Object.keys(gain).length) return; | |
| $("#cacheResult").hidden = false; | |
| const labels = { | |
| added_special_dense: "dense special tokens", | |
| agentic_swe: "agentic coding", | |
| cmn_Hani: "Chinese", | |
| code_mixed: "source code", | |
| eng_Latn: "English web text", | |
| math_latex: "math and LaTeX", | |
| prefix_sharing: "shared prompt prefix", | |
| }; | |
| const action = $("#cacheResult").dataset.inspectLabel; | |
| const rows = Object.entries(gain).sort((a, b) => b[1] - a[1]) | |
| .map(([name, value]) => ({ | |
| name: labels[name] || name, v: value, us: value >= 1, | |
| inspect: "cachePreview", previewKey: name, action, | |
| })); | |
| $("#cacheChart").innerHTML = bars(rows, | |
| {unit: "× throughput with cache enabled", el: $("#cacheChart")}); | |
| const triggers = [...$("#cacheChart").querySelectorAll("[data-inspect='cachePreview']")]; | |
| const preview = $("#cachePreview"); | |
| const segments = $("#cachePreviewSegments"); | |
| const note = $("#cachePreviewNote"); | |
| const close = preview.querySelector(".input-preview-close"); | |
| let activeTrigger = null; | |
| const showInput = trigger => { | |
| const key = trigger.dataset.previewKey; | |
| const input = D.cache_previews?.[key]; | |
| if (!input) return; | |
| segments.replaceChildren(...input.segments.map(segment => { | |
| const box = document.createElement("div"); | |
| box.className = `input-segment ${segment.kind}`; | |
| const label = document.createElement("div"); | |
| label.className = "input-segment-label"; | |
| label.textContent = $("#cacheResult").dataset[`${segment.kind}Label`]; | |
| const text = document.createElement("pre"); | |
| text.textContent = segment.text; | |
| box.append(label, text); | |
| return box; | |
| })); | |
| note.textContent = $("#cacheResult").dataset[ | |
| key === "prefix_sharing" ? "prefixNote" : "corpusNote" | |
| ]; | |
| }; | |
| const setOpen = (open, trigger = activeTrigger) => { | |
| preview.hidden = !open; | |
| triggers.forEach(item => item.setAttribute("aria-expanded", String(open && item === trigger))); | |
| if (open) { activeTrigger = trigger; showInput(trigger); } | |
| }; | |
| triggers.forEach(trigger => { | |
| trigger.onclick = () => setOpen(!(activeTrigger === trigger && !preview.hidden), trigger); | |
| trigger.onkeydown = event => { | |
| if (event.key !== "Enter" && event.key !== " ") return; | |
| event.preventDefault(); | |
| setOpen(!(activeTrigger === trigger && !preview.hidden), trigger); | |
| }; | |
| }); | |
| close.onclick = () => { setOpen(false); activeTrigger?.focus(); }; | |
| preview.onkeydown = event => { | |
| if (event.key === "Escape") { setOpen(false); activeTrigger?.focus(); } | |
| }; | |
| } | |
| renderCache(); | |
| /* ---- tables ----------------------------------------------------------- */ | |
| function table(el, cols, rows){ | |
| $(el).innerHTML = "<thead><tr>" + cols.map(c => `<th>${esc(c)}</th>`).join("") + "</tr></thead><tbody>" | |
| + rows.map(r => "<tr>" + r.map(c => { | |
| const [v, cl] = Array.isArray(c) ? c : [c, ""]; | |
| // {h: "..."} is pre-built markup; anything else is escaped text | |
| return `<td class="${cl}">${v && v.h !== undefined ? v.h : esc(v)}</td>`; | |
| }).join("") + "</tr>").join("") + "</tbody>"; | |
| } | |
| const L = D.latency; | |
| if (L) table("#tLat", ["", "0.23 p50", "v1 p50", "0.23 p99", "v1 p99", "p99 faster"], | |
| L.rows.map(r => [r.name, | |
| r.rel_p50.toFixed(1) + " \u00b5s", [r.new_p50.toFixed(2) + " \u00b5s", "a"], | |
| r.rel_p99.toFixed(1) + " \u00b5s", [r.new_p99.toFixed(2) + " \u00b5s", "a"], | |
| [(r.rel_p99 / r.new_p99).toFixed(0) + "\u00d7", "k"]])); | |
| function renderMemory(){ | |
| const M = D.memory_modes?.[selectedMemoryMode]; | |
| if (!M) return; | |
| const unsupported = r => { | |
| const why = r.explanation || r.status || "This engine cannot expose this configuration."; | |
| return {h: `<span class="unsupported" tabindex="0" title="${attr(why)}"` | |
| + ` aria-label="unsupported: ${attr(why)}">unsupported</span>`}; | |
| }; | |
| table("#tMemory", ["engine", "loaded heap", "heap after warm encode"], M.rows.map(r => [ | |
| r.name, | |
| r.loaded === null ? unsupported(r) : r.loaded.toFixed(1) + " MB", | |
| r.working === null ? unsupported(r) : [r.working.toFixed(1) + " MB", r.us ? "a" : ""], | |
| ])); | |
| } | |
| renderMemory(); | |
| /* ---- linked crate footprint ------------------------------------------ */ | |
| const crateSelection = {serialize: false, convert: false, train: false}; | |
| const featureSelection = { | |
| unigram: false, wordpiece: false, wordlevel: false, normalizers: false, | |
| "unicode-scripts": false, parallelism: false, | |
| }; | |
| let selectedFeatureProfile = "minsize"; | |
| function renderCrates(){ | |
| const C = D.crate_sizes; | |
| if (!C) return; | |
| const labels = { | |
| serialize: ["tk-serialize", "read tokenizer.json"], | |
| convert: ["tk-convert", "upgrade legacy JSON"], | |
| train: ["tk-train", "train vocabularies"], | |
| }; | |
| const featureLabels = { | |
| unigram: ["unigram", "Unigram model"], | |
| wordpiece: ["wordpiece", "WordPiece model"], | |
| wordlevel: ["wordlevel", "WordLevel model"], | |
| normalizers: ["normalizers", "Unicode normalization"], | |
| "unicode-scripts": ["unicode-scripts", "script splitting"], | |
| parallelism: ["parallelism", "parallel batches"], | |
| }; | |
| if (!$("#cratePick").children.length){ | |
| $("#cratePick").innerHTML = `<div class="crate-node">` | |
| + `<label class="required"><input type="checkbox" checked disabled>` | |
| + `<span class="cn">tk-encode</span><span class="cd">required runtime</span></label>` | |
| + `<div class="crate-features"><label class="crate-profile">build profile ` | |
| + `<select id="featureProfileSelect"><option value="minsize">minsize</option>` | |
| + `<option value="slimest">slimest</option></select></label>` | |
| + `<label class="required"><input type="checkbox" checked disabled>` | |
| + `<span class="cn">bpe</span><span class="cd">included in tk-encode</span></label>` | |
| + C.feature_options.map(k => `<label><input type="checkbox" data-feature="${k}">` | |
| + `<span class="cn">${featureLabels[k][0]}</span>` | |
| + `<span class="cd">${featureLabels[k][1]}` | |
| + `<span class="delta" data-feature-delta="${k}"></span></span></label>`).join("") | |
| + `<div class="feature-linked"><span id="featureCurrentName"></span>` | |
| + `<strong id="featureCurrentSize"></strong></div></div></div>` | |
| + C.options.map(k => `<label><input type="checkbox" data-crate="${k}"` | |
| + `${crateSelection[k] ? " checked" : ""}><span class="cn">${labels[k][0]}</span>` | |
| + `<span class="cd">${labels[k][1]}` | |
| + `<span class="delta" data-crate-delta="${k}"></span></span></label>`).join(""); | |
| $("#cratePick").querySelectorAll("input[data-crate]").forEach(input => { | |
| input.onchange = () => { crateSelection[input.dataset.crate] = input.checked; renderCrates(); }; | |
| }); | |
| $("#cratePick").querySelectorAll("input[data-feature]").forEach(input => { | |
| input.onchange = () => { | |
| featureSelection[input.dataset.feature] = input.checked; | |
| renderCrates(); | |
| }; | |
| }); | |
| $("#featureProfileSelect").onchange = event => { | |
| selectedFeatureProfile = event.target.value; | |
| renderCrates(); | |
| }; | |
| } | |
| const enabled = C.options.filter(k => crateSelection[k]); | |
| const key = enabled.join("+") || "encode"; | |
| const featureKey = C.feature_options.filter(k => featureSelection[k]).join("+") || "bpe"; | |
| const featureSizes = C.feature_profiles[selectedFeatureProfile]; | |
| const featureBytes = featureSizes[featureKey]; | |
| const optionalDelta = C.configs[key] - C.configs.encode; | |
| const bytes = featureBytes + optionalDelta; | |
| const kb = n => (n / 1000).toFixed(1) + " kB"; | |
| const setToggleDelta = (element, delta) => { | |
| element.textContent = ` · ${delta < 0 ? "−" : "+"}${kb(Math.abs(delta))}`; | |
| element.classList.toggle("add", delta >= 0); | |
| element.classList.toggle("remove", delta < 0); | |
| }; | |
| C.options.forEach(option => { | |
| const toggled = C.options.filter(name => | |
| name === option ? !crateSelection[name] : crateSelection[name]); | |
| const toggledKey = toggled.join("+") || "encode"; | |
| const delta = C.configs[toggledKey] - C.configs[key]; | |
| setToggleDelta($(`[data-crate-delta="${option}"]`), delta); | |
| }); | |
| C.feature_options.forEach(option => { | |
| const toggled = C.feature_options.filter(name => | |
| name === option ? !featureSelection[name] : featureSelection[name]); | |
| const toggledKey = toggled.join("+") || "bpe"; | |
| const delta = featureSizes[toggledKey] - featureBytes; | |
| setToggleDelta($(`[data-feature-delta="${option}"]`), delta); | |
| }); | |
| $("#crateOldName").textContent = C.baseline.name; | |
| $("#crateOldSize").textContent = kb(C.baseline.bytes); | |
| $("#crateCurrentName").textContent = enabled.length | |
| ? `selected crates · ${selectedFeatureProfile} estimate` | |
| : `tk-encode · ${selectedFeatureProfile}`; | |
| $("#crateCurrentSize").textContent = kb(bytes); | |
| $("#featureCurrentName").textContent = `tk-encode · ${selectedFeatureProfile}`; | |
| $("#featureCurrentSize").textContent = kb(featureBytes); | |
| } | |
| renderCrates(); | |
| // The five stages, with what changed in each. Columns are computed rather than | |
| // eyeballed: box centres drive where the annotations hang. Each box is its own | |
| // span so a stage can be filled solid while it is doing the work. | |
| (function(){ | |
| const ST = [{n: "frame"}, {n: "normalize"}, {n: "split"}, | |
| {n: "model"}, {n: "post"}]; | |
| const SEP = "──▶"; | |
| const tops = [], mids = [], bots = [], centre = []; | |
| let col = 0; | |
| ST.forEach((st, i) => { | |
| const w = st.n.length + 2; | |
| centre.push(col + 1 + Math.floor(w / 2)); | |
| // border glyphs are their own span: when a stage is filled they take the fill | |
| // colour and disappear, so the block reads solid instead of as a nested box | |
| const e = t => `<span class="e">${t}</span>`; | |
| tops.push(`<span class="bx b${i}">${e("┌" + "─".repeat(w) + "┐")}</span>`); | |
| mids.push(`<span class="bx b${i}">${e("│ ")}<span class="s">${st.n}</span>${e(" │")}</span>`); | |
| bots.push(`<span class="bx b${i}">${e("└" + "─".repeat(w) + "┘")}</span>`); | |
| col += w + 2 + (i < ST.length - 1 ? 3 : 0); | |
| }); | |
| const place = items => { | |
| let out = ""; | |
| for (const [c, txt] of items.sort((x, y) => x[0] - y[0])) | |
| out += " ".repeat(Math.max(0, c - out.length)) + txt; | |
| return out; | |
| }; | |
| const marked = ST.map((st, i) => [st, centre[i]]).filter(([st]) => st.a); | |
| const caret = place(marked.map(([, c]) => [c, "▲"])); | |
| const order = marked.slice().reverse(); | |
| const rows = order.map(([st, c], k) => | |
| place(order.slice(k + 1).map(([, c2]) => [c2, "│"]).concat([[c, "└── " + st.a]]))); | |
| let nums = place(ST.map((st, i) => [centre[i], String(i)])); | |
| ST.forEach((st, i) => { | |
| nums = nums.replace(new RegExp(`(^|\\s)${i}(?=\\s|$)`), | |
| (m0, sp) => sp + `<span class="num n${i}">${i}</span>`); | |
| }); | |
| $("#pipe").innerHTML = [ | |
| tops.join(" "), | |
| mids.join(SEP) + SEP + ` <span class="m">token ids</span>`, | |
| bots.join(" "), | |
| nums, | |
| `<span class="a">${esc(caret)}</span>`, | |
| ...rows.map(r => `<span class="a">${esc(r)}</span>`), | |
| ].join("\n"); | |
| // ---- walk one real sentence through the stages ------------------------- | |
| // Each frame wipes in from the left over the one before it, so the reader sees | |
| // the transformation happen rather than a set of before/after states. No single | |
| // model exercises every stage, so the model is switchable and each one shows its | |
| // own real behaviour. | |
| const DM = D.demo || []; | |
| if (!DM.length) return; | |
| const chips = a => a.map(t => esc(t)).join(`<span class="sep">│</span>`); | |
| const plainOf = a => a.join("│"); | |
| function frames(M){ | |
| const nDesc = M.nkind === "lower" | |
| ? `${M.ntype} lowercases the text before anything else sees it` | |
| : M.nkind === "codepoints" | |
| ? `${M.ntype} rewrites the codepoints before anything else sees them` | |
| : `${M.m} defines no normalizer, so nothing changes here`; | |
| const sDesc = "cut into pre-tokens. merges never cross these boundaries"; | |
| const mDesc = M.marker === "##" | |
| ? "merge with WordPiece rules. ## marks a continuation" | |
| : "merge each pre-token with the learned rules"; | |
| const pre = (M.addpre || []).join(" "), post = (M.addpost || []).join(" "); | |
| const pDesc = pre && post ? `wraps the sequence in ${pre} … ${post}` | |
| : pre ? `prepends ${pre}, the extra id at the front` | |
| : post ? `appends ${post}, the extra id at the end` | |
| : `${M.m} adds no special tokens, so these ids are final`; | |
| return [ | |
| {box: 0, name: "frame", desc: "the raw bytes arrive", p: M.x, h: esc(M.x)}, | |
| {box: 1, name: "normalize", desc: nDesc, p: M.n, h: esc(M.n)}, | |
| {box: 2, name: "split", desc: sDesc, p: plainOf(M.b), h: chips(M.b)}, | |
| {box: 3, name: "model", desc: mDesc, p: plainOf(M.t), h: chips(M.t)}, | |
| {box: 3, name: "model", desc: "look every token up in the vocabulary", p: plainOf(M.ib), h: chips(M.ib)}, | |
| {box: 4, name: "post", desc: pDesc, p: plainOf(M.i), h: chips(M.i)}, | |
| ]; | |
| } | |
| const WIPE = 10, HOLD = 12; | |
| function buildWalk(M){ | |
| const F = frames(M), seq = []; | |
| for (let k = 0; k < F.length; k++){ | |
| const cur = F[k], prev = F[(k - 1 + F.length) % F.length]; | |
| const L = Math.max(prev.p.length, cur.p.length); | |
| for (let t = 1; t <= WIPE; t++){ | |
| const n = Math.round(t / WIPE * L); | |
| seq.push({...cur, body: esc(cur.p.slice(0, n)) + `<span class="cur">▌</span>` | |
| + esc(prev.p.slice(Math.min(n, prev.p.length)))}); | |
| } | |
| for (let h = 0; h < HOLD; h++) seq.push({...cur, body: cur.h}); | |
| } | |
| return seq; | |
| } | |
| let mi = 0, seq = buildWalk(DM[0]), tick = 0; | |
| const light = k => { | |
| $("#pipe").querySelectorAll(".bx,.num").forEach(e => e.classList.remove("on")); | |
| $("#pipe").querySelectorAll(`.b${k},.n${k}`).forEach(e => e.classList.add("on")); | |
| }; | |
| $("#mtabs").innerHTML = DM.map((d, i) => | |
| `<button role="tab" data-i="${i}" aria-selected="${i === 0}">${esc(d.m)}</button>`).join(""); | |
| const mbtns = [...$("#mtabs").querySelectorAll("button")]; | |
| mbtns.forEach((btn, i) => { | |
| btn.onclick = () => { | |
| mi = i; seq = buildWalk(DM[i]); tick = 0; | |
| mbtns.forEach((b2, j) => b2.setAttribute("aria-selected", j === i)); | |
| }; | |
| }); | |
| // tick is kept here rather than taken from the toggler's counter because the | |
| // model tabs reset it: switching model restarts the walk at its first frame. | |
| toggler("a-pipe", (t, dir) => { | |
| tick = ((tick + dir) % seq.length + seq.length) % seq.length; | |
| const f = seq[tick]; | |
| light(f.box); | |
| $("#walk").innerHTML = `<span class="k">${f.box} · ${f.name}</span>` | |
| + `<span class="d">${" ".repeat(Math.max(1, 14 - f.name.length))}${esc(f.desc)}</span>\n` | |
| + `<span class="v">${f.body}</span>`; | |
| }, 22); | |
| })(); | |
| // Each change links to the section that takes it apart. The displayed number is | |
| // read back off that section's heading rather than written here, so renumbering | |
| // the page cannot leave this table pointing at the wrong one. | |
| const GOTO = { | |
| "no-alloc model": "merge", "bitcannon": "split", "merge-loop rewrite": "merge", | |
| "workspace split": "ranking", | |
| "word cache": "cache", "native parallelism": "ranking", | |
| }; | |
| const secNum = id => { | |
| const n = document.querySelector(`#${id} h2 .n`); | |
| return n ? n.textContent.trim() : ""; | |
| }; | |
| const chg = name => { | |
| const id = GOTO[name]; | |
| return [{h: `<a href="#${id}">${esc(name)}</a>` | |
| + `<span class="go">→ ${esc(secNum(id))}</span>`}, "k"]; | |
| }; | |
| table("#tChanges", ["change", "what it does"], [ | |
| [chg("workspace split"), [{h: "one crate became a workspace, measured in the crate size tab " | |
| + 'above: <span class="hi">tk-encode</span> is the required runtime, and ' | |
| + '<span class="hi">tk-serialize</span>, <span class="hi">tk-convert</span> and ' | |
| + '<span class="hi">tk-train</span> are linked only when an application needs them'}, "d"]], | |
| [chg("no-alloc model"), ["the merge working set lives in a caller-owned scratch buffer; the " | |
| + "loop never touches the allocator", "d"]], | |
| [chg("bitcannon"), ["the split pattern becomes Boolean operations over bitstreams, " | |
| + "using SIMD instructions to find splits instead of a regex engine", "d"]], | |
| [chg("merge-loop rewrite"), ["the pieces being merged form an intrusive doubly-linked list " | |
| + "inside one preallocated buffer, so a merge updates two indices instead of " | |
| + "moving data", "d"]], | |
| [chg("word cache"), ["a thread-local memo from pre-token bytes to finished ids, so a repeated " | |
| + "word is merged once", "d"]], | |
| [chg("native parallelism"), [{h: "one shared tokenizer encodes from many threads at once; " | |
| + "each thread draws its scratch buffer and word cache from its own sub-pool, so " | |
| + "threads no longer queue on a single lock " | |
| + '(<a href="https://github.com/huggingface/tokenizers/pull/2365">#2365</a>)'}, "d"]], | |
| ]); | |
| table("#tMethod", ["rule", "why"], [ | |
| [["one timing loop", "k"], ["every engine runs the identical loop; no per-engine fast path", "d"]], | |
| [["load excluded", "k"], ["vocabulary load is timed separately, never inside encode", "d"]], | |
| [["id-hash verified", "k"], ["FNV-1a over the output ids must match the baseline exactly", "d"]], | |
| [["common cells only", "k"], ["medians are over cells every engine ran and verified", "d"]], | |
| [["complete sweep per process", "k"], ["each repeat starts in a new process and retains every cell", "d"]], | |
| [["physical-core pinning", "k"], ["workers are pinned to eight distinct physical cores, never sibling SMT threads", "d"]], | |
| [["independent Jobs", "k"], ["separate Jobs measure host-to-host variation", "d"]], | |
| ]); | |
| /* ---- absolute rates as a line chart ------------------------------------- | |
| Threads are evenly spaced (a category axis, not a numeric one) so the shape | |
| of each curve is legible; the y axis is linear from zero, which is what makes | |
| the released library read as flat rather than merely slower. */ | |
| function renderThreadLines(){ | |
| const view = D.scaling_modes?.[selectedScalingMode]; | |
| const t = view?.mt || D.tabs.find(t => t.id === "mt"); | |
| if (!t) return; | |
| const ths = t.threads || [1, 2, 4, 8]; | |
| // averaged over every language and model, not one arbitrary pair | |
| const all = view?.mt_all || D.mt_all || {}; | |
| const series = ["tokenizers v1", "tokenizers 0.23"].map(n => { | |
| const row = t.rows.find(r => r.name === n); | |
| if (!row) return null; | |
| const key = n === "tokenizers v1" ? "pipeline" : "hf-tokenizers"; | |
| const c = all[key]; | |
| return c ? {...row, curve: c.map(p => Math.round(p.mbps))} : row; | |
| }).filter(Boolean); | |
| if (!series.length) return; | |
| const max = Math.max(...series.flatMap(r => r.curve)); | |
| const H = 13; | |
| const YW = String(max).length + 1; // y-label gutter | |
| const W = Math.max(28, fitW($("#mtLine"), YW + 20)); | |
| const xs = ths.map((_, i) => Math.round(i * (W - 1) / (ths.length - 1))); | |
| const grid = Array.from({length: H}, () => Array(W).fill(" ")); | |
| const cls = Array.from({length: H}, () => Array(W).fill("")); | |
| const yOf = v => H - 1 - Math.round(v / max * (H - 1)); | |
| series.forEach(r => { | |
| const g = r.us ? BLK : MED; | |
| const k = r.us ? "us" : "ref"; | |
| for (let i = 0; i < r.curve.length; i++){ | |
| // interpolate to the next point so the series reads as a line | |
| if (i < r.curve.length - 1){ | |
| const x0 = xs[i], x1 = xs[i + 1], y0 = yOf(r.curve[i]), y1 = yOf(r.curve[i + 1]); | |
| for (let x = x0; x <= x1; x++){ | |
| const y = Math.round(y0 + (y1 - y0) * (x - x0) / Math.max(1, x1 - x0)); | |
| grid[y][x] = g; cls[y][x] = k; | |
| } | |
| } | |
| const y = yOf(r.curve[i]); | |
| grid[y][xs[i]] = g; cls[y][xs[i]] = k; | |
| } | |
| }); | |
| // emit, coalescing runs of the same class into one span | |
| const line = (row, crow) => { | |
| let out = "", i = 0; | |
| while (i < W){ | |
| let j = i; while (j < W && crow[j] === crow[i]) j++; | |
| const txt = esc(row.slice(i, j).join("")); | |
| out += crow[i] ? `<span class="bar ${crow[i]}">${txt}</span>` : txt; | |
| i = j; | |
| } | |
| return out; | |
| }; | |
| const yLab = new Set([0, H - 1]); | |
| let out = `<span class="row"><span class="x">${pad("", YW)}throughput, MB/s · linear from zero` | |
| + `</span></span>`; | |
| for (let y = 0; y < H; y++){ | |
| // label only the rows a series ends on, so the gutter stays quiet | |
| let lab = ""; | |
| for (const r of series) | |
| if (yOf(r.curve[r.curve.length - 1]) === y) lab = String(r.curve[r.curve.length - 1]); | |
| out += `<span class="row"><span class="x">${lpad(lab, YW - 1)} </span>` | |
| + `<span class="dim">┤</span>` + line(grid[y], cls[y]) + `</span>`; | |
| } | |
| // its own axis row, so a flat series never merges into the axis | |
| out += `<span class="row"><span class="x">${lpad("0", YW - 1)} </span>` | |
| + `<span class="dim">└${"─".repeat(W)}</span></span>`; | |
| // x axis | |
| let ax = Array(W).fill(" "); | |
| ths.forEach((n, i) => { const t2 = String(n); for (let c = 0; c < t2.length; c++){ | |
| const x = Math.min(W - 1, Math.max(0, xs[i] - (i === ths.length - 1 ? t2.length - 1 : 0) + c)); | |
| ax[x] = t2[c]; } }); | |
| out += `<span class="row"><span class="x">${pad("", YW)}</span><span class="dim">` | |
| + esc(ax.join("")) + ` threads</span></span>`; | |
| out += `<span class="row"><span class="x">${pad("", YW + 1)}</span>` | |
| + series.map(r => `<span class="lab${r.us ? " us" : " ref"}">${r.us ? BLK : MED} ` | |
| + `${esc(r.name)}</span>`) | |
| .join(`<span class="x"> </span>`) + `</span>`; | |
| $("#mtLine").innerHTML = out; | |
| renderThreadTable(); | |
| } | |
| /* ---- multithread rates, every engine ----------------------------------- */ | |
| function renderThreadTable(){ | |
| const view = D.scaling_modes?.[selectedScalingMode]; | |
| const t = view?.mt || D.tabs.find(t => t.id === "mt"); | |
| if (!t) return; | |
| const ths = t.threads || [1, 2, 4, 8]; | |
| const all = view?.mt_all || D.mt_all || {}; | |
| const NAME = {pipeline: "tokenizers v1", "hf-tokenizers": "tokenizers 0.23"}; | |
| const rows = Object.entries(all).map(([e, c]) => ({ | |
| name: NAME[e] || e, | |
| curve: c.map(p => Math.round(p.mbps)), | |
| eff: Math.round(c[c.length - 1].efficiency_pct | |
| ?? 100 * c[c.length - 1].mbps / (c[0].mbps * ths[ths.length - 1])), | |
| us: e === "pipeline", ref: e === "hf-tokenizers", | |
| })).sort((a, b) => b.curve[b.curve.length - 1] - a.curve[a.curve.length - 1]); | |
| const labw = Math.max(...rows.map(r => r.name.length)) + 1; | |
| let out = `<span class="row"><span class="lab dim">${pad("", labw)}</span><span class="x">${ | |
| ths.map(n => lpad(n + "t", 8)).join("")} efficiency</span></span>\n`; | |
| for (const r of rows){ | |
| const cls = r.us ? " us" : (r.ref ? " ref" : ""); | |
| out += `<span class="row"><span class="lab${cls}">${esc(pad(r.name, labw))}</span>` | |
| + `<span class="val${r.us ? " us" : ""}">${r.curve.map(v => lpad(String(v), 8)).join("")}</span>` | |
| + `<span class="x"> ${lpad(r.eff + "%", 6)}</span></span>\n`; | |
| } | |
| $("#mtChart").innerHTML = out; | |
| } | |
| renderThreadTable(); | |
| /* hero: a scanline tokenizing a field of text */ | |
| (function(){ | |
| // Verbatim from the tokenizers docs, docs/source/pipeline.rst. The masthead | |
| // scrolls the description of the pipeline the rest of this page takes apart. | |
| const SRC = "When calling Tokenizer.encode or Tokenizer.encode_batch, the input text(s) go through " | |
| + "the following pipeline: Normalization is, in a nutshell, a set of operations you apply " | |
| + "to a raw string to make it less random or \"cleaner\". Once the input texts are normalized" | |
| + " and pre-tokenized, the Tokenizer applies the model on the pre-tokens. The role of the " | |
| + "model is to split your \"words\" into tokens, using the rules it has learned. It's also " | |
| + "responsible for mapping those tokens to their corresponding IDs in the vocabulary of the" | |
| + " model. Post-processing is the last step of the tokenization pipeline, to perform any " | |
| + "additional transformation to the Encoding before it's returned, like adding potential " | |
| + "special tokens. "; | |
| const el = $("#field"); | |
| let COLS = 0, ROWS = 14, lines = []; | |
| function build(){ | |
| COLS = Math.max(40, Math.min(150, Math.floor(el.clientWidth / (el.getBoundingClientRect().width / 150 || 7)))); | |
| COLS = Math.max(40, Math.floor(window.innerWidth * 0.92 / Math.max(4, parseFloat(getComputedStyle(el).fontSize) * 0.6))); | |
| lines = []; | |
| for (let r = 0; r < ROWS; r++){ | |
| let s = ""; | |
| const off = (r * 37) % SRC.length; | |
| while (s.length < COLS + 2) s += SRC.slice(off) + SRC.slice(0, off); | |
| lines.push(s.slice(0, COLS)); | |
| } | |
| } | |
| build(); | |
| addEventListener("resize", build); | |
| toggler("field", t => { | |
| const head = (t * 3) % (COLS + 30); | |
| let out = ""; | |
| for (let r = 0; r < ROWS; r++){ | |
| const src = lines[r], skew = head - r * 2; | |
| let row = ""; | |
| for (let c = 0; c < COLS; c++){ | |
| const ch = src[c] || " "; | |
| if (c > skew) row += `<span>${esc(ch)}</span>`; | |
| else if (c > skew - 4) row += `<span class="s">${esc(ch === " " ? "·" : ch.toUpperCase())}</span>`; | |
| else if (ch === " ") row += `<span class="b">│</span>`; | |
| else row += `<span class="t">${esc(ch)}</span>`; | |
| } | |
| out += row + "\n"; | |
| } | |
| el.innerHTML = out; | |
| }, 14); | |
| })(); | |
| /* split race: regex byte-by-byte vs bitcannon a register at a time */ | |
| (function(){ | |
| const S = "tokenizers v1 splits text with bitstreams, 64 bytes at a time."; | |
| const el = $("#splitPre"); | |
| const bounds = new Set(); for (let i = 0; i < S.length; i++) if (S[i] === " ") bounds.add(i); | |
| function line(name, done, note){ | |
| let s = ""; | |
| for (let i = 0; i < S.length; i++){ | |
| if (i < done) s += bounds.has(i) ? `<span class="g">│</span>` : `<span class="w">${esc(S[i])}</span>`; | |
| else s += `<span class="m">${esc(S[i])}</span>`; | |
| } | |
| return `<span class="m">${pad(name, 12)}</span>${s} <span class="g">${note}</span>\n`; | |
| } | |
| toggler("a-split", t => { | |
| const N = S.length, cyc = N + 14; | |
| const re = Math.min(N, (t % cyc)); | |
| const at = Math.min(N, Math.floor(t % cyc) * 64); | |
| el.innerHTML = line("regex", re, re >= N ? "done, " + N + (N === 1 ? " step" : " steps") : re + " bytes") | |
| + line("bitcannon", at, at >= N ? (n => "done, " + n + (n === 1 ? " step" : " steps"))(Math.ceil(N / 64)) : at + " bytes"); | |
| }, 12); | |
| })(); | |
| /* cache stream: a hit is a lookup, a miss is the whole merge loop. The bar is | |
| the work each word costs. That is the entire point of the cache, so it is | |
| what the animation shows. Schematic: the ratio is roughly what the measured | |
| stage costs imply, not a trace. */ | |
| (function(){ | |
| const rnd = () => (seedv = (seedv * 1103515245 + 12345) & 0x7fffffff) / 0x7fffffff; | |
| const COMMON = ("the of and to in a is that for it as was with be by on not he this are or his from at " | |
| + "which but have an they one you had word tokenizers pipeline encode merge cache thread scratch").split(" "); | |
| const ZIPF = []; | |
| COMMON.forEach((w, i) => { for (let k = 0; k < Math.ceil(60 / (i + 1)); k++) ZIPF.push(w); }); | |
| // Heaps' law is the whole point of the section: a corpus never stops producing | |
| // words it has not seen before, it just produces them more slowly. A fixed word | |
| // list would go 100% hit after warmup and show nothing, so the stream keeps | |
| // minting novel words at roughly the miss rate measured on English. | |
| const RARE = ("quantisation isotope verbatim chromatography polymorphic Reykjavik unwieldy " | |
| + "photolysis heuristic bramble Kaganovich sublinear thaumaturgy epoxide Wollongong " | |
| + "misattributed glyphset zeolite paraffin Anantapur tessellate rubidium hagiography " | |
| + "unmarshalling Trebizond cuneiform xenolith pellucid Zwolle interstitial obsidian " | |
| + "recrudescent Ljubljana palimpsest gyroscope Nakhchivan effloresce quinquennial " | |
| + "bathymetry Uttaradit susurrus lodestar Þingvellir apophenia kudzu Mahabalipuram " | |
| + "tremolite winnow Oaxaca peristalsis bezoar Kaliningrad orthography feldspar " | |
| + "Ouagadougou vermiculite tessera Yekaterinburg sarsaparilla").split(" "); | |
| const NOVEL = 0.12; // schematic stream, not benchmark data | |
| let rareI = 0; | |
| const nextWord = () => rnd() < NOVEL | |
| ? (rareI < RARE.length ? RARE[rareI++] : "term" + (rareI++)) | |
| : ZIPF[Math.floor(rnd() * ZIPF.length)]; | |
| const el = $("#cachePre"); | |
| const HIT_T = 2, MISS_T = 14, BARW = MISS_T, KEEP = 6; | |
| const seen = new Set(); | |
| let tape = [], cur = null, done = 0, uncached = 0, seedv = 7; | |
| // Join the stream already warm. A cache that starts empty misses on everything | |
| // for its first few hundred words, which is true but is not the case worth | |
| // showing. The steady state is what a real corpus spends almost all its time in. | |
| // The warm-up must not consume the rare pool, or the visible stream runs out | |
| // of real words and starts inventing them. | |
| for (let i = 0; i < 400; i++) { | |
| if (rnd() < NOVEL) done += MISS_T; | |
| else { const w = ZIPF[Math.floor(rnd() * ZIPF.length)]; | |
| done += seen.has(w) ? HIT_T : (seen.add(w), MISS_T); } | |
| uncached += MISS_T; | |
| } | |
| const bar = (n, cls) => `<span class="${cls}">${BLK.repeat(n)}</span>` | |
| + `<span class="m">${LGT.repeat(Math.max(0, BARW - n))}</span>`; | |
| const row = (e, n, live) => | |
| `<span class="${e.hit ? "hit" : "miss"}">${pad(e.hit ? "HIT" : "MISS", 5)}</span>` | |
| + `<span class="${live ? "w" : "m"}">${esc(pad("Ġ" + e.w, 15))}</span>` | |
| + bar(n, e.hit ? "hit" : "w") | |
| + `<span class="m"> ${e.hit ? "lookup" : (live ? "merging…" : "merged")}</span>`; | |
| toggler("a-cache", () => { | |
| if (!cur) { | |
| const w = nextWord(); | |
| const hit = seen.has(w); | |
| if (!hit) seen.add(w); | |
| cur = {w: w, hit: hit, need: hit ? HIT_T : MISS_T, at: 0}; | |
| } | |
| cur.at++; | |
| let live = cur; | |
| if (cur.at >= cur.need) { | |
| tape.unshift(cur); tape = tape.slice(0, KEEP); | |
| done += cur.need; uncached += MISS_T; | |
| cur = null; | |
| } | |
| const saved = uncached ? done / uncached : 1; | |
| const sw = Math.round(saved * 28); | |
| el.innerHTML = | |
| `<span class="m">${pad("", 5)}${pad("word", 15)}work to produce its ids</span>\n` | |
| + (live ? row(live, live.at, true) + "\n" : "") | |
| + tape.slice(0, KEEP - (live ? 1 : 0)).map(e => row(e, e.need, false)).join("\n") | |
| + `\n\n<span class="m">work done </span>` | |
| + `<span class="hit">${BLK.repeat(sw)}</span><span class="m">${LGT.repeat(28 - sw)}</span>` | |
| + `<span class="w"> ${(100 * saved).toFixed(0)}%</span>` | |
| + `<span class="m"> of uncached work</span>`; | |
| }, 16); | |
| })(); | |
| /* ---- headings tokenise themselves -------------------------------------- | |
| Frames come from make_titles.py, which runs the real gpt2 tokenizer over | |
| every heading on this page: pre-tokenize, byte-level, merge, ids. It then | |
| asserts the ids decode back to the heading. The animation walks out to ids | |
| and back, so it ends on exactly the text it started from. */ | |
| (function(){ | |
| const TT = D.titles || {}; | |
| if (reduced || !Object.keys(TT).length) return; | |
| const SEP = `<span class="sep">│</span>`; | |
| const join = (parts, cls) => parts | |
| .map(t => cls ? `<span class="${cls}">${esc(t)}</span>` : esc(t)).join(SEP); | |
| function framesFor(spec){ | |
| const f = [ | |
| [spec.x, ""], [spec.p.join("│"), "p"], [spec.b.join("│"), "b"], | |
| [spec.t.join("│"), "t"], [spec.i.join("│"), "i"], | |
| [spec.t.join("│"), "t"], [spec.b.join("│"), "b"], [spec.x, ""], | |
| ]; | |
| // consecutive stages are often identical (one pre-token, one token). | |
| // holding the same frame twice reads as a stall, so collapse them | |
| return f.filter((x, k) => k === 0 || x[0] !== f[k - 1][0]); | |
| } | |
| const html = (spec, kind) => kind === "i" ? join(spec.i, "id") | |
| : kind === "t" ? join(spec.t) : kind === "b" ? join(spec.b) | |
| : kind === "p" ? join(spec.p) : esc(spec.x); | |
| const HOLD = 9, WIPE = 6, MS = 34; | |
| // The whole animation is precomputed as a flat list of rendered frames, so | |
| // stepping it is a single index and there is no state to get wrong mid-run. | |
| function buildSeq(spec){ | |
| const F = framesFor(spec), seq = []; | |
| for (let k = 0; k < F.length; k++){ | |
| if (k > 0){ | |
| const prev = F[k - 1][0], cur = F[k][0]; | |
| const L = Math.max(prev.length, cur.length); | |
| for (let t = 1; t <= WIPE; t++){ | |
| const n = Math.round(t / WIPE * L); | |
| seq.push(esc(cur.slice(0, n)) + `<span class="sep">▌</span>` | |
| + esc(prev.slice(Math.min(n, prev.length)))); | |
| } | |
| } | |
| const rendered = html(spec, F[k][1]); | |
| for (let h = 0; h < HOLD; h++) seq.push(rendered); | |
| } | |
| return seq; | |
| } | |
| function run(el, spec){ | |
| if (el.dataset.busy) return; | |
| el.dataset.busy = "1"; | |
| el.classList.add("on"); | |
| const seq = el._seq || (el._seq = buildSeq(spec)); | |
| let i = 0; | |
| const timer = setInterval(() => { | |
| if (i >= seq.length){ | |
| clearInterval(timer); | |
| el.innerHTML = esc(spec.x); // always rest on the heading itself | |
| el.classList.remove("on"); | |
| delete el.dataset.busy; | |
| return; | |
| } | |
| el.innerHTML = seq[i++]; | |
| }, MS); | |
| } | |
| const targets = []; | |
| document.querySelectorAll("h2 > span:nth-of-type(2), h3").forEach(el => { | |
| const spec = TT[el.textContent.trim()]; | |
| if (!spec) return; | |
| el.classList.add("tk"); | |
| el.title = "tokenise"; | |
| el.onclick = () => run(el, spec); | |
| targets.push([el, spec]); | |
| }); | |
| // fire once when a heading first scrolls into view, staggered so the page | |
| // never has more than a couple of them moving at once | |
| const io = new IntersectionObserver(es => { | |
| es.filter(e => e.isIntersecting).forEach((e, n) => { | |
| const hit = targets.find(([el]) => el === e.target); | |
| if (hit) setTimeout(() => run(hit[0], hit[1]), n * 260); | |
| io.unobserve(e.target); | |
| }); | |
| }, {threshold: 1, rootMargin: "0px 0px -12% 0px"}); | |
| targets.forEach(([el]) => io.observe(el)); | |
| })(); | |
| /* ---- contents rail ------------------------------------------------------- | |
| Built from the sections themselves rather than a hand-kept list, so adding a | |
| section to the template puts it in the rail automatically. The current | |
| section is the last one whose top has passed the reading line. */ | |
| (function(){ | |
| const rail = $("#rail"); | |
| const secs = [...document.querySelectorAll("section[id]")]; | |
| if (!rail || !secs.length) return; | |
| rail.innerHTML = `<span class="top">contents</span><ol>` + secs.map(sec => { | |
| const h = sec.querySelector("h2"); | |
| const n = h?.querySelector(".n")?.textContent.trim() || ""; | |
| const t = h?.querySelector("span:nth-of-type(2)")?.textContent.trim() || sec.id; | |
| const sub = n.includes(".") ? " sub" : ""; | |
| return `<li><a class="${sub}" href="#${sec.id}">` | |
| + `<span class="n">${esc(n)}</span>${esc(t)}</a></li>`; | |
| }).join("") + `</ol>`; | |
| const links = new Map(secs.map(sec => | |
| [sec.id, rail.querySelector(`a[href="#${sec.id}"]`)])); | |
| let current = null; | |
| function mark(){ | |
| // the reading line sits a third of the way down the viewport | |
| const line = innerHeight * 0.33; | |
| let id = secs[0].id; | |
| for (const sec of secs) if (sec.getBoundingClientRect().top <= line) id = sec.id; | |
| // near the very bottom the last section may never cross the line | |
| if (innerHeight + scrollY >= document.body.scrollHeight - 4) id = secs[secs.length - 1].id; | |
| if (id === current) return; | |
| if (current) links.get(current)?.removeAttribute("aria-current"); | |
| links.get(id)?.setAttribute("aria-current", "true"); | |
| current = id; | |
| } | |
| let tick = false; | |
| addEventListener("scroll", () => { | |
| if (tick) return; | |
| tick = true; | |
| requestAnimationFrame(() => { mark(); tick = false; }); | |
| }, {passive: true}); | |
| addEventListener("resize", mark, {passive: true}); | |
| mark(); | |
| })(); | |
| /* ---- "run this in tokbench" --------------------------------------------- | |
| These assume a prepared checkout with the tokbench binary on PATH. An | |
| omitted model or corpus means every available value in that dimension. */ | |
| (function(){ | |
| const CMD = { | |
| lead: () => "tokbench measure encode --engine all", | |
| latency: () => "tokbench measure latency --engine pipeline --compare-to hf-tokenizers --corpus eng_Latn", | |
| decode: () => "tokbench measure decode --engine pipeline --compare-to hf-tokenizers", | |
| architecture: () => "tokbench measure encode --engine pipeline --compare-to hf-tokenizers", | |
| memory: () => { | |
| const view = D.memory_modes[selectedMemoryMode]; | |
| return "tokbench measure memory --engine all --model gpt-oss --corpus eng_Latn --threads " | |
| + view.threads + " --scaling-mode " + view.parallelism + " --reps 3"; | |
| }, | |
| "crate-size": () => "tokbench measure crate-size", | |
| threads: () => "tokbench measure scaling --engine all --corpus eng_Latn --corpus cmn_Hani --max-threads 8 --scaling-mode " + selectedScalingMode, | |
| cache: () => "tokbench measure encode --engine pipeline --engine hf-tokenizers --compare-to pipeline-no-cache", | |
| }; | |
| document.querySelectorAll("[data-repro]").forEach(host => { | |
| const kind = host.dataset.repro; | |
| const command = () => CMD[kind](); | |
| const btn = document.createElement("button"); | |
| btn.className = "repro-btn"; | |
| btn.setAttribute("aria-expanded", "false"); | |
| btn.textContent = "run this in tokbench"; | |
| const panel = document.createElement("div"); | |
| panel.className = "repro-panel"; | |
| panel.innerHTML = `<div class="hd"><span>reproduce</span>` | |
| + `<button class="cp">copy</button></div><pre>${esc(command())}</pre>`; | |
| host.appendChild(btn); | |
| host.appendChild(panel); | |
| btn.onclick = () => { | |
| panel.querySelector("pre").textContent = command(); | |
| const open = panel.classList.toggle("on"); | |
| btn.setAttribute("aria-expanded", String(open)); | |
| }; | |
| const cp = panel.querySelector(".cp"); | |
| cp.onclick = async () => { | |
| try { | |
| await navigator.clipboard.writeText(command()); | |
| cp.textContent = "copied"; | |
| } catch { | |
| // clipboard is often blocked in an embedded frame; select it instead | |
| const r = document.createRange(); | |
| r.selectNodeContents(panel.querySelector("pre")); | |
| const sel = getSelection(); | |
| sel.removeAllRanges(); sel.addRange(r); | |
| cp.textContent = "selected, press copy"; | |
| } | |
| setTimeout(() => { cp.textContent = "copy"; }, 2000); | |
| }; | |
| }); | |
| })(); | |
| function drawAll(){ measureCh(); renderChart(); renderDecode(); renderCache(); renderThreadLines(); renderKeyWins(); } | |
| drawAll(); | |
| let rt; | |
| addEventListener("resize", () => { clearTimeout(rt); rt = setTimeout(drawAll, 120); }); | |
| </script> | |
| <style id="embed"> | |
| :root, :root[data-theme="dark"]{ | |
| --bg:#0b0f19; --ink:#c9ced8; --hi:#f3f4f6; --dim:#8b93a7; | |
| --rule:#1f2937; --ref:#4b5563; | |
| } | |
| @media (prefers-color-scheme:light){ | |
| :root{ --bg:#ffffff; --ink:#374151; --hi:#111827; --dim:#6b7280; | |
| --rule:#e5e7eb; --ref:#9ca3af; } | |
| } | |
| :root[data-theme="light"]{ | |
| --bg:#ffffff; --ink:#374151; --hi:#111827; --dim:#6b7280; | |
| --rule:#e5e7eb; --ref:#9ca3af; | |
| } | |
| /* The blog gives the iframe a fixed height and its sanitizer rewrites the tag, | |
| so scrolling="no" cannot be relied on. A few pixels of reflow at an unlucky | |
| width would otherwise raise a scrollbar inside the frame, on top of the | |
| blog's own. HEIGHTS in make_blog_post.py carries the slack that makes | |
| hiding the overflow safe rather than lossy. | |
| Hiding it unconditionally forced every box to be sized for the worst case, | |
| a 320px phone, which left a desktop reader staring at hundreds of pixels of | |
| blank. So the boxes are sized for the blog's reading column instead, and | |
| below that width the figure scrolls rather than clipping: a scrollbar on a | |
| phone is honest, silent truncation is not. */ | |
| html, body{overflow:hidden} | |
| @media (max-width:640px){ html, body{overflow:auto} } | |
| /* body's background normally propagates to the canvas. Paint it directly so | |
| the slack below the content does not depend on that rule. */ | |
| html{background:var(--bg)} | |
| /* the rail is a reading aid for a standalone page; in a frame it has nowhere | |
| to go */ | |
| #rail{display:none } | |
| .key-win h3{text-transform:capitalize} | |
| /* Both hf.co/blog and community articles frame these with | |
| sandbox="allow-scripts", which withholds allow-popups and | |
| allow-top-navigation, so no link inside a figure can open anywhere. Keep the | |
| words, drop the affordance: an underline that does nothing is worse than | |
| plain text. The prose in the post carries the real links. */ | |
| a{color:inherit; border-bottom:0; pointer-events:none; cursor:text} | |
| a:hover{color:inherit; border-bottom:0} | |
| /* embed mode: the blog supplies the page, this supplies one figure */ | |
| .wrap{max-width:none; margin:0} | |
| header, #rail{display:none } | |
| section{display:none ; margin:0 } | |
| section > h2{display:none } | |
| body{padding:1.1rem 1.2rem} | |
| /* the page animates a hero field behind the masthead; nothing to animate here */ | |
| #field{display:none } | |
| section#split{display:block } | |
| #split > p:not(.note){display:none } | |
| #a-split{margin-top:0} | |
| </style> | |
| </body> | |
| </html> | |