Spaces:
Running
Running
Update page, card, Apache-2.0 text and NOTICE
Browse files- LICENSE-Apache-2.0 +202 -0
- NOTICE +27 -0
- README.md +27 -9
- index.html +29 -20
- style.css +0 -28
LICENSE-Apache-2.0
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright [yyyy] [name of copyright owner]
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
NOTICE
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
The WGSL source embedded in index.html (the KERNEL constant, shown on the page
|
| 2 |
+
as "Kernel 0112 source") is a WebGPU shader emitted at run time by the npm
|
| 3 |
+
package @litert-lm/core, version 0.17.1, published by Google LLC
|
| 4 |
+
(https://www.npmjs.com/package/@litert-lm/core, repository
|
| 5 |
+
https://github.com/google-ai-edge/LiteRT-LM). It was recorded through the
|
| 6 |
+
WebGPU API (createShaderModule) during the capture run described in the repository README while the package loaded the
|
| 7 |
+
model bundle gemma-4-E2B-it-web.litertlm in Chrome on an Apple M2 Max, by the
|
| 8 |
+
harness in https://github.com/abgnydn/litert-capture (capture.mjs,
|
| 9 |
+
www/capture.html), where it is out/shaders/0112_unlabeled.wgsl. The text in
|
| 10 |
+
index.html is the captured text with trailing spaces removed from three lines
|
| 11 |
+
and is otherwise unmodified. The number 0112 is its creation order; the
|
| 12 |
+
package assigns no labels. The page derives its variants from this text at
|
| 13 |
+
run time.
|
| 14 |
+
|
| 15 |
+
The package declares the Apache License, Version 2.0 (npm "license" field;
|
| 16 |
+
the repository carries the same license) but ships no LICENSE or NOTICE file
|
| 17 |
+
in its tarball. The license text is reproduced in the LICENSE-Apache-2.0 file next to
|
| 18 |
+
this notice, as section 4 of the license asks. Any copyright in the shader
|
| 19 |
+
is Google LLC's. No claim of endorsement by Google is made; this Space
|
| 20 |
+
is independent work.
|
| 21 |
+
|
| 22 |
+
This shader is treated here as covered by the package's Apache-2.0 license.
|
| 23 |
+
The package hands it to the browser's createShaderModule as plain text
|
| 24 |
+
whenever it runs, so this Space republishes nothing that a user of the
|
| 25 |
+
package cannot already read. If Google considers this redistribution outside
|
| 26 |
+
that license, open an issue at https://github.com/abgnydn/litert-capture/issues
|
| 27 |
+
and the shader will be removed.
|
README.md
CHANGED
|
@@ -11,8 +11,6 @@ tags:
|
|
| 11 |
- webgpu
|
| 12 |
- benchmark
|
| 13 |
- gpu
|
| 14 |
-
models:
|
| 15 |
-
- litert-community/gemma-4-E2B-it-litert-lm
|
| 16 |
---
|
| 17 |
|
| 18 |
# Kernel 0112
|
|
@@ -21,21 +19,41 @@ One of the 153 GPU programs Google's LiteRT-LM web engine (`@litert-lm/core` 0.1
|
|
| 21 |
compiles to run Gemma 4 in the browser. This page runs it on your GPU as shipped,
|
| 22 |
then with a bigger workgroup (a control) and with each dot product split 16 and 32
|
| 23 |
ways, and checks every variant against a CPU reference. On an Apple M2 Max the
|
| 24 |
-
32-way split is 1.65x faster in Chrome 146
|
| 25 |
-
|
| 26 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
1.53x, from a single run of an f32 transcription of the kernel, copied by hand from
|
| 28 |
the notebook output, because Chrome exposed no 16-bit float shaders there. In the
|
| 29 |
-
running model on the M2 Max,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
|
| 31 |
Needs WebGPU with 16-bit float shaders (`shader-f16`). Tested in Chrome. The page
|
| 32 |
checks on load and tells you if your browser or GPU does not offer it. No model
|
| 33 |
weights and no dataset are downloaded; the page loads its fonts from Google
|
| 34 |
-
Fonts. No result leaves your browser; copy the JSON if you want to share it.
|
| 35 |
|
| 36 |
Method, harness and full record: https://github.com/abgnydn/litert-capture
|
| 37 |
|
| 38 |
The kernel source shown on the page was emitted by `@litert-lm/core` (Google LLC,
|
| 39 |
Apache License 2.0) and is reproduced with attribution; it is not covered by the
|
| 40 |
-
MIT license declared above for this Space.
|
| 41 |
-
|
|
|
|
|
|
| 11 |
- webgpu
|
| 12 |
- benchmark
|
| 13 |
- gpu
|
|
|
|
|
|
|
| 14 |
---
|
| 15 |
|
| 16 |
# Kernel 0112
|
|
|
|
| 19 |
compiles to run Gemma 4 in the browser. This page runs it on your GPU as shipped,
|
| 20 |
then with a bigger workgroup (a control) and with each dot product split 16 and 32
|
| 21 |
ways, and checks every variant against a CPU reference. On an Apple M2 Max the
|
| 22 |
+
32-way split is 1.65x faster in Chrome 146, a browser inferred from the
|
| 23 |
+
adapter string (the two records behind that figure predate the provenance
|
| 24 |
+
field and carry only the adapter string `apple`/`metal-3`; the
|
| 25 |
+
provenanced `out/microbench-0112-2026-09-24T07-36-19-655Z.json` gives 1.638x for
|
| 26 |
+
the same kernel on the same machine with `provenance.browser`
|
| 27 |
+
`Chrome/146.0.7680.153`), and in Chrome 131 on the same machine the
|
| 28 |
+
single-dispatch harness measures the split as slower (0.84x, one run); both records
|
| 29 |
+
are committed. On a Colab T4 it is
|
| 30 |
1.53x, from a single run of an f32 transcription of the kernel, copied by hand from
|
| 31 |
the notebook output, because Chrome exposed no 16-bit float shaders there. In the
|
| 32 |
+
running model on the M2 Max, with this kernel and the three other quantized
|
| 33 |
+
matrix-vector kernels split deeper, decode goes from 57 to 71 tokens per second, 1.20x
|
| 34 |
+
as the median of the per-repetition pairings (1.18 to 1.33x) and 1.25x as the
|
| 35 |
+
ratio of the condition medians. Leaving one kernel out at a time puts the
|
| 36 |
+
largest marginal contributions within the four-kernel patch on the two 4-bit
|
| 37 |
+
kernels 0099 and 0105 and smaller ones on this one and 0113, which cannot be
|
| 38 |
+
ordered against each other because which of them ranks last follows from the
|
| 39 |
+
choice of aggregate. Leaving this one out of the four-kernel patch costs 0.240
|
| 40 |
+
ms per token by the condition medians, while patched alone, in an earlier
|
| 41 |
+
two-repetition sweep with a fixed condition order (`out/patch.json`, not
|
| 42 |
+
directly comparable), it saved 0.690 and 0.885 ms per token, near the roughly
|
| 43 |
+
0.96 ms its isolated gain predicts over its 40 dispatches per token; with the
|
| 44 |
+
other three patched its marginal contribution is smaller than its
|
| 45 |
+
stand-alone saving in that earlier sweep; whether that reflects overlapping
|
| 46 |
+
gains or the difference between the two sweeps is not established.
|
| 47 |
|
| 48 |
Needs WebGPU with 16-bit float shaders (`shader-f16`). Tested in Chrome. The page
|
| 49 |
checks on load and tells you if your browser or GPU does not offer it. No model
|
| 50 |
weights and no dataset are downloaded; the page loads its fonts from Google
|
| 51 |
+
Fonts. No result leaves your browser; copy the JSON if you want to share it. The copied JSON contains your browser's user-agent string and GPU name.
|
| 52 |
|
| 53 |
Method, harness and full record: https://github.com/abgnydn/litert-capture
|
| 54 |
|
| 55 |
The kernel source shown on the page was emitted by `@litert-lm/core` (Google LLC,
|
| 56 |
Apache License 2.0) and is reproduced with attribution; it is not covered by the
|
| 57 |
+
MIT license declared above for this Space. The Apache-2.0 license text is in
|
| 58 |
+
`LICENSE-Apache-2.0` and the kernel's provenance in `NOTICE`, both in this Space, and the page links the
|
| 59 |
+
license. Independent work, not affiliated with Google.
|
index.html
CHANGED
|
@@ -127,31 +127,32 @@
|
|
| 127 |
<section class="wide" id="results">
|
| 128 |
<div class="panel" id="mine" hidden>
|
| 129 |
<div class="panel-head"><span class="title">Your GPU</span><span class="meta" id="mine-meta"></span></div>
|
| 130 |
-
<div class="headline"><span class="big" id="mine-speedup">–</span><span class="what">faster when each dot product is split 32 ways instead of 4, same 64-thread workgroups, and the same answer to within 16-bit float rounding</span></div>
|
| 131 |
<div class="rows" id="mine-rows"></div>
|
| 132 |
<div class="legend"><span class="o">original kernel</span><span>changed kernel</span><span class="h">same, weights already in cache (hot)</span></div>
|
| 133 |
<div class="controls" style="margin-top:16px"><button class="quiet" id="copy" type="button">Copy results as JSON</button><span class="status" id="copy-status" style="margin:0"></span></div>
|
|
|
|
| 134 |
</div>
|
| 135 |
<div class="panel" id="ref">
|
| 136 |
-
<div class="panel-head"><span class="title">Reference run</span><span class="meta">Apple M2 Max (30-core GPU) · Chrome 146 · 2026-09-22 · node harness · one timestamp per run · 576 cold and 24 hot samples · 24 matrices</span></div>
|
| 137 |
-
<div class="headline"><span class="big">1.65×</span><span class="what">faster when each dot product is split 32 ways instead of 4, same 64-thread workgroups, and the same answer to within 16-bit float rounding</span></div>
|
| 138 |
<div class="rows" id="ref-rows"></div>
|
| 139 |
<div class="legend"><span class="o">original kernel</span><span>changed kernel</span><span class="h">same, weights already in cache (hot)</span></div>
|
| 140 |
-
<p class="note">Mean of two harness runs' medians, each variant timed forward and again in reverse order after a discarded warm-up, with the browser's timestamp rounding switched off.
|
| 141 |
</div>
|
| 142 |
</section>
|
| 143 |
|
| 144 |
<main class="wrap">
|
| 145 |
<h2>What is being measured</h2>
|
| 146 |
-
<p>Generating one token with a language model means multiplying the model's stored numbers, its <em>weights</em>, by the current input, layer after layer. This kernel does one such multiplication: a matrix of 12,288 × 1,536 weights, each stored in 2 bits, times a vector of 1,536 inputs. That is 4.5
|
| 147 |
-
<p>Because the work is dominated by reading those 4.5
|
| 148 |
-
<p>To read 4.5
|
| 149 |
|
| 150 |
<figure>
|
| 151 |
<svg id="diagram" viewBox="-48 0 770 390" role="img" aria-labelledby="diagram-title">
|
| 152 |
<title id="diagram-title">Matrix of 12,288 by 1,536 two-bit weights multiplied by a 1,536 input vector gives 12,288 outputs; below, the original 16 by 4 workgroup, a 64 by 4 control with the same total threads, and the 2 by 32 workgroup that dispatches eight times as many threads.</title>
|
| 153 |
</svg>
|
| 154 |
-
<figcaption>Top: the multiplication this kernel performs. Bottom: the work is split into <em>workgroups</em>, small teams of GPU threads. Left, the original: teams of 16 × 4, four threads per output slice, 12,288 threads in total. Middle, the control: teams of 64 × 4, four times bigger, still 12,288 threads; it runs no faster. Right, the change: teams of 2 × 32, still 64 threads each, but 32 threads per output slice and 98,304 threads in total; it runs about 1.65× faster on the reference machine. The 2 × 32 shape is the one Google's own kernel 0113 already uses.</figcaption>
|
| 155 |
</figure>
|
| 156 |
|
| 157 |
<h3>The four variants</h3>
|
|
@@ -167,17 +168,17 @@
|
|
| 167 |
|
| 168 |
<h2>How it is measured</h2>
|
| 169 |
<ol class="steps">
|
| 170 |
-
<li><strong>Random weights, made here.</strong> The page generates 24 different 4.5
|
| 171 |
<li><strong>A reference answer on the CPU.</strong> For the first matrix, plain JavaScript computes the 12,288 outputs. Every GPU variant must match it to within 16-bit float rounding (0.05 on outputs of size about 2) or it is marked wrong, and a wrong variant's speed does not count.</li>
|
| 172 |
-
<li><strong>Cold and hot.</strong> "Cold" rotates through all 24 matrices (108
|
| 173 |
<li><strong>GPU timestamps, in batches.</strong> The GPU records the time before and after each batch of 48 back-to-back runs. Browsers may round these timestamps to 100 µs for privacy; batching keeps that rounding between about 3% of the slowest bar and 6% of the fastest. Where timestamps are unavailable the page times batches of 480 runs from JavaScript instead and says so in the results panel.</li>
|
| 174 |
-
<li><strong>Interleaved order after a warm-up.</strong> The GPU is slow for whatever is measured first.
|
| 175 |
</ol>
|
| 176 |
|
| 177 |
<h3>What the numbers mean</h3>
|
| 178 |
<ul class="notes">
|
| 179 |
<li><strong>µs per run</strong> is the GPU time of a batch of 48 runs divided by 48, then the median over sixteen batches. Lower is better.</li>
|
| 180 |
-
<li><strong>GB/s</strong> is the 4,718,592 bytes of weights (4.5 MiB) divided by that time, in decimal gigabytes
|
| 181 |
<li><strong>×</strong> is the original kernel's cold time divided by the 32-way split's cold time.</li>
|
| 182 |
<li>A <span class="badge ok">matches reference</span> badge means the variant produced the same outputs as the CPU within rounding. If a variant shows <span class="badge no">wrong answer</span>, its speed does not count.</li>
|
| 183 |
</ul>
|
|
@@ -190,8 +191,9 @@
|
|
| 190 |
</details>
|
| 191 |
|
| 192 |
<h2>What the kernel costs in a whole token</h2>
|
| 193 |
-
<p>Across a whole generated token (one traced run and one 7-token profile run), nine weight-reading kernels move about 741 MiB
|
| 194 |
-
<p>
|
|
|
|
| 195 |
<p>This page's reference numbers are from an Apple M2 Max; the repository README holds the Colab T4 run. Copy yours with the button above.</p>
|
| 196 |
|
| 197 |
<footer>Independent work, not affiliated with Google. Kernel source emitted by <code>@litert-lm/core</code> 0.17.1, published by Google under the <a href="https://www.apache.org/licenses/LICENSE-2.0" target="_blank" rel="noopener noreferrer">Apache License 2.0</a>. Measurement code and text by the page's author. Method, harness and full record: <a href="https://github.com/abgnydn/litert-capture" target="_blank" rel="noopener noreferrer">github.com/abgnydn/litert-capture</a>.</footer>
|
|
@@ -203,7 +205,7 @@
|
|
| 203 |
const OUT_SLICES = 3072, IN_SLICES = 384
|
| 204 |
const WEIGHT_BYTES = OUT_SLICES * IN_SLICES * 4
|
| 205 |
const BATCH = 48, ROUNDS = 8
|
| 206 |
-
// 24 matrices = 108
|
| 207 |
// measurable share of 'cold' reads warm. Falls back to 8 on GPU out-of-memory,
|
| 208 |
// and the results panel then says so.
|
| 209 |
let NMAT = 24
|
|
@@ -216,7 +218,11 @@
|
|
| 216 |
{ key: 'split32', label: '32-way split', sub: '2 × 32 workgroup · 98,304 threads', ks: 32, wgX: 2 },
|
| 217 |
]
|
| 218 |
const HEADLINE = ['orig', 'split32']
|
| 219 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 220 |
orig: { cold: 60.9, hot: 44.8, err: 6.25e-3 }, wg256: { cold: 60.6, hot: 46.6, err: 6.25e-3 },
|
| 221 |
split16: { cold: 40.2, hot: 33.0, err: 3.44e-3 }, split32: { cold: 36.8, hot: 34.5, err: 2.04e-3 },
|
| 222 |
}
|
|
@@ -335,6 +341,7 @@ fn main(@builtin(global_invocation_id) reserved_gid : vec3<u32>,
|
|
| 335 |
|
| 336 |
// ---------- small helpers ----------
|
| 337 |
const $ = (id) => document.getElementById(id)
|
|
|
|
| 338 |
const gbs = (us) => WEIGHT_BYTES / (us * 1e-6) / 1e9
|
| 339 |
const median = (a) => { const s = [...a].sort((x, y) => x - y); return s[Math.floor(s.length / 2)] }
|
| 340 |
const f16 = (x) => { const f = new Float32Array([x]), u = new Uint32Array(f.buffer)[0]; const s = (u >> 16) & 0x8000, e = ((u >> 23) & 0xff) - 112, m = u & 0x7fffff; if (e <= 0) return s; if (e >= 31) return s | 0x7c00; let h = s | (e << 10) | (m >> 13); if ((m & 0x1fff) > 0x1000 || ((m & 0x1fff) === 0x1000 && (h & 1))) h++; return h }
|
|
@@ -374,7 +381,7 @@ fn main(@builtin(global_invocation_id) reserved_gid : vec3<u32>,
|
|
| 374 |
const text = (x, y, s, attrs = {}) => { const t = add('text', { x, y, fill: 'var(--muted)', 'font-family': 'var(--mono)', 'font-size': 11, ...attrs }); t.textContent = s; return t }
|
| 375 |
add('rect', { x: 40, y: 22, width: 240, height: 96, rx: 3, fill: 'var(--orig-soft)', stroke: 'var(--orig)' })
|
| 376 |
text(160, 66, 'weights', { 'text-anchor': 'middle', fill: 'var(--ink)', 'font-family': 'var(--body)', 'font-size': 13, 'font-weight': 600 })
|
| 377 |
-
text(160, 84, '12,288 × 1,536 · 2-bit · 4.5
|
| 378 |
text(160, 14, '1,536 inputs wide', { 'text-anchor': 'middle' })
|
| 379 |
text(30, 70, '12,288 rows', { 'text-anchor': 'end' })
|
| 380 |
text(300, 74, '×', { fill: 'var(--ink)', 'font-size': 18 })
|
|
@@ -430,7 +437,7 @@ fn main(@builtin(global_invocation_id) reserved_gid : vec3<u32>,
|
|
| 430 |
let gpuError = null
|
| 431 |
device.addEventListener('uncapturederror', (e) => { gpuError = String(e.error?.message ?? e) })
|
| 432 |
|
| 433 |
-
say(`Generating ${NMAT} random weight matrices (${(NMAT * WEIGHT_BYTES / 1048576).toFixed(0)}
|
| 434 |
const weights = []
|
| 435 |
for (let m = 0; m < NMAT; m++) {
|
| 436 |
const a = new Uint8Array(WEIGHT_BYTES)
|
|
@@ -545,7 +552,9 @@ fn main(@builtin(global_invocation_id) reserved_gid : vec3<u32>,
|
|
| 545 |
// present
|
| 546 |
lastResult = { page: 'kernel-0112', version: 2, kernel: '@litert-lm/core 0.17.1 shader 0112', date: new Date().toISOString(), gpu: adapterInfo, userAgent: navigator.userAgent, timing: hasTs ? (quantized ? 'gpu-timestamps-quantized-100us' : 'gpu-timestamps') : 'cpu-batch', batch: B, rounds: ROUNDS, positions: 2, matrices: NMAT, weightBytes: WEIGHT_BYTES, variants: Object.fromEntries(VARIANTS.map((v) => [v.key, { ks: v.ks, wgX: v.wgX, threads: OUT_SLICES * v.ks }])), results }
|
| 547 |
const speed = results[HEADLINE[0]].cold / results[HEADLINE[1]].cold
|
| 548 |
-
|
|
|
|
|
|
|
| 549 |
$('mine-meta').textContent = `${[adapterInfo.vendor, adapterInfo.architecture].filter(Boolean).join(' · ') || 'this GPU'} · ${lastResult.timing.replace(/-/g, ' ')} · ${B * ROUNDS * 2} runs per number · ${NMAT} matrices${NMAT < 24 ? ' (reduced: GPU memory)' : ''}`
|
| 550 |
scaleMax = niceMax(Math.max(scaleMax, ...Object.values(results).map((d) => gbs(d.hot))))
|
| 551 |
renderRows($('mine-rows'), results, scaleMax); renderRows($('ref-rows'), REFERENCE, scaleMax)
|
|
@@ -556,7 +565,7 @@ fn main(@builtin(global_invocation_id) reserved_gid : vec3<u32>,
|
|
| 556 |
device.destroy()
|
| 557 |
} catch (e) {
|
| 558 |
const msg = String(e?.message ?? e)
|
| 559 |
-
if (/out of memory|allocation|createBuffer|createTexture/i.test(msg) && NMAT > 8) { say(`Not enough GPU memory for ${NMAT} matrices; retrying with 8 (36
|
| 560 |
say('The measurement stopped: ' + msg)
|
| 561 |
}
|
| 562 |
btn.disabled = false
|
|
|
|
| 127 |
<section class="wide" id="results">
|
| 128 |
<div class="panel" id="mine" hidden>
|
| 129 |
<div class="panel-head"><span class="title">Your GPU</span><span class="meta" id="mine-meta"></span></div>
|
| 130 |
+
<div class="headline"><span class="big" id="mine-speedup">–</span><span class="what" id="mine-what">faster when each dot product is split 32 ways instead of 4, same 64-thread workgroups, and the same answer to within 16-bit float rounding</span></div>
|
| 131 |
<div class="rows" id="mine-rows"></div>
|
| 132 |
<div class="legend"><span class="o">original kernel</span><span>changed kernel</span><span class="h">same, weights already in cache (hot)</span></div>
|
| 133 |
<div class="controls" style="margin-top:16px"><button class="quiet" id="copy" type="button">Copy results as JSON</button><span class="status" id="copy-status" style="margin:0"></span></div>
|
| 134 |
+
<p class="note">The copied JSON includes your browser's user-agent string and GPU name.</p>
|
| 135 |
</div>
|
| 136 |
<div class="panel" id="ref">
|
| 137 |
+
<div class="panel-head"><span class="title">Reference run</span><span class="meta">Apple M2 Max (30-core GPU) · Chrome 146 (inferred) · 2026-09-22 · node harness · one timestamp per run · 576 cold and 24 hot samples · 24 matrices</span></div>
|
| 138 |
+
<div class="headline"><span class="big">1.65×</span><span class="what">faster when each dot product is split 32 ways instead of 4, same 64-thread workgroups, and the same answer to within 16-bit float rounding; the same harness reads 0.84× on Chrome 131</span></div>
|
| 139 |
<div class="rows" id="ref-rows"></div>
|
| 140 |
<div class="legend"><span class="o">original kernel</span><span>changed kernel</span><span class="h">same, weights already in cache (hot)</span></div>
|
| 141 |
+
<p class="note">Mean of two harness runs' medians, each variant timed forward and again in reverse order after a discarded warm-up, with the browser's timestamp rounding switched off. At the original's thread count the bigger-workgroup control reads the same as the original (60.6 against 60.9 µs), so workgroup size alone does not move the original's number; at 49,152 threads the repository README records a small secondary workgroup-size effect. Chrome 146 is inferred for these two records, which carry only the adapter string <code>apple</code>/<code>metal-3</code> and no browser field; a later single-dispatch record of the same kernel on the same machine, <code>out/microbench-0112-2026-09-24T07-36-19-655Z.json</code>, gives 1.638× with its browser recorded as <code>Chrome/146.0.7680.153</code>. The node harness behind these numbers times one dispatch at a time; on Chrome for Testing 131 on the same machine it reports the reverse in a committed record, 78.1 µs for the original against 93.1 µs for the 32-way split, while this page's batched method still favours the split there, which is a console observation with no committed record. On the reference machine this page's own method lands above or below the harness figure from session to session, and on a contended or battery-powered GPU it reads lower, below 1.0× in one reading on a throttled machine on battery, also a console observation with no committed record, with every variant still matching the reference; the repository README records the range.</p>
|
| 142 |
</div>
|
| 143 |
</section>
|
| 144 |
|
| 145 |
<main class="wrap">
|
| 146 |
<h2>What is being measured</h2>
|
| 147 |
+
<p>Generating one token with a language model means multiplying the model's stored numbers, its <em>weights</em>, by the current input, layer after layer. This kernel does one such multiplication: a matrix of 12,288 × 1,536 weights, each stored in 2 bits, times a vector of 1,536 inputs. That is 4.5 MiB of weights the kernel reads each time it runs, and it runs 40 times per token. Byte counts on this page are MiB (2<sup>20</sup> bytes) and GB/s is decimal; figures quoted from Google's model card are the card's own MB.</p>
|
| 148 |
+
<p>Because the work is dominated by reading those 4.5 MiB, the yardstick used here is <strong>effective weight-streaming bandwidth</strong>: the bytes of weights the kernel reads per unit of kernel time, shown as GB/s, which is not a measurement of memory traffic, since a byte served from cache counts the same. For scale, the Apple M2 Max is <a href="https://www.apple.com/newsroom/2023/01/apple-unveils-m2-pro-and-m2-max-next-generation-chips-for-next-level-workflows/" target="_blank" rel="noopener noreferrer">rated at 400 GB/s</a>. The fastest kernel measured in this project, a 4-bit one in a single run, reaches 195 GB/s, about half of that. The original kernel reaches about 77 GB/s here (60.9 µs, the two-run mean cold median), and 80 GB/s inside the running model (one 7-token profile run).</p>
|
| 149 |
+
<p>To read 4.5 MiB the kernel launches 12,288 threads: four for each group of four outputs, each thread computing those four outputs over a quarter of the 1,536 inputs. That is few threads for a chip with thousands of execution lanes, which typically hides memory latency by keeping many reads in flight. The variants below change one thing at a time: the size of the thread teams (workgroups) the engine uses for this shape, and how many threads share each dot product.</p>
|
| 150 |
|
| 151 |
<figure>
|
| 152 |
<svg id="diagram" viewBox="-48 0 770 390" role="img" aria-labelledby="diagram-title">
|
| 153 |
<title id="diagram-title">Matrix of 12,288 by 1,536 two-bit weights multiplied by a 1,536 input vector gives 12,288 outputs; below, the original 16 by 4 workgroup, a 64 by 4 control with the same total threads, and the 2 by 32 workgroup that dispatches eight times as many threads.</title>
|
| 154 |
</svg>
|
| 155 |
+
<figcaption>Top: the multiplication this kernel performs. Bottom: the work is split into <em>workgroups</em>, small teams of GPU threads. Left, the original: teams of 16 × 4, four threads per output slice, 12,288 threads in total. Middle, the control: teams of 64 × 4, four times bigger, still 12,288 threads; it runs no faster. Right, the change: teams of 2 × 32, still 64 threads each, but 32 threads per output slice and 98,304 threads in total; it runs about 1.65× faster on the reference machine in Chrome 146 (inferred from the adapter string), and slower in a single-dispatch run on Chrome 131. The 2 × 32 shape is the one Google's own kernel 0113 already uses on the transposed shape (12,288 → 1,536): the engine uses 16 × 4 for this shape and 2 × 32 for the transposed one, so what separates 0112 from the faster configuration on this adapter in Chrome 146 is the configuration chosen for its shape; how the engine chooses is not established.</figcaption>
|
| 156 |
</figure>
|
| 157 |
|
| 158 |
<h3>The four variants</h3>
|
|
|
|
| 168 |
|
| 169 |
<h2>How it is measured</h2>
|
| 170 |
<ol class="steps">
|
| 171 |
+
<li><strong>Random weights, made here.</strong> The page generates 24 different 4.5 MiB matrices of random bytes in your browser (108 MiB; if the GPU cannot hold them it retries with 8 and the results panel says so). No model weights are downloaded and no results leave your browser; the page's fonts are fetched from Google's font servers. The real model's weights are not needed to measure speed: a memory-bound kernel should take the same time whatever the values, and the bytes are unpatterned because some GPUs compress patterned textures.</li>
|
| 172 |
<li><strong>A reference answer on the CPU.</strong> For the first matrix, plain JavaScript computes the 12,288 outputs. Every GPU variant must match it to within 16-bit float rounding (0.05 on outputs of size about 2) or it is marked wrong, and a wrong variant's speed does not count.</li>
|
| 173 |
+
<li><strong>Cold and hot.</strong> "Cold" rotates through all 24 matrices (108 MiB, far more than the reference chip's caches hold), which makes cache reuse from one run to the next unlikely, as in the real model where 40 different matrices stream past. "Hot" reuses one matrix, small enough to stay in the chip's cache. The gap between them shows how much a variant is limited by memory rather than by arithmetic.</li>
|
| 174 |
<li><strong>GPU timestamps, in batches.</strong> The GPU records the time before and after each batch of 48 back-to-back runs. Browsers may round these timestamps to 100 µs for privacy; batching keeps that rounding between about 3% of the slowest bar and 6% of the fastest. Where timestamps are unavailable the page times batches of 480 runs from JavaScript instead and says so in the results panel.</li>
|
| 175 |
+
<li><strong>Interleaved order after a warm-up.</strong> The GPU is slow for whatever is measured first. A discarded warm-up of the original kernel (16 batches) runs before any timing, then every variant is timed in order and again in reverse, and the two positions' samples are pooled: eight batches per variant per position, median reported. Other tabs, thermal state and power mode all move the numbers; run twice if the first looks odd.</li>
|
| 176 |
</ol>
|
| 177 |
|
| 178 |
<h3>What the numbers mean</h3>
|
| 179 |
<ul class="notes">
|
| 180 |
<li><strong>µs per run</strong> is the GPU time of a batch of 48 runs divided by 48, then the median over sixteen batches. Lower is better.</li>
|
| 181 |
+
<li><strong>GB/s</strong> is effective weight-streaming bandwidth: the 4,718,592 bytes of weights (4.5 MiB) divided by that time, in decimal gigabytes, counting a byte the same whether it was read from memory or served from cache. Higher is better; your chip's rated memory bandwidth is the scale to read it against.</li>
|
| 182 |
<li><strong>×</strong> is the original kernel's cold time divided by the 32-way split's cold time.</li>
|
| 183 |
<li>A <span class="badge ok">matches reference</span> badge means the variant produced the same outputs as the CPU within rounding. If a variant shows <span class="badge no">wrong answer</span>, its speed does not count.</li>
|
| 184 |
</ul>
|
|
|
|
| 191 |
</details>
|
| 192 |
|
| 193 |
<h2>What the kernel costs in a whole token</h2>
|
| 194 |
+
<p>Across a whole generated token (one traced run and one 7-token profile run), nine weight-reading kernels move about 741 MiB in 277 runs. On the reference machine that is an effective weight-streaming bandwidth of 81 GB/s as an overhead-adjusted estimate (78 GB/s raw; the per-kernel measurement moves each dispatch into its own timestamped pass, which adds about 1.5 µs per pass, and the adjusted figure subtracts it), a fifth of the <a href="https://www.apple.com/newsroom/2023/01/apple-unveils-m2-pro-and-m2-max-next-generation-chips-for-next-level-workflows/" target="_blank" rel="noopener noreferrer">400 GB/s</a> the chip is rated for. Four of them, this one included, take 48.5% of the GPU time, and each launches only 6,144 to 12,288 threads for 4.5 MiB.</p>
|
| 195 |
+
<p>Deeper K splits were then applied inside the running model on the reference machine to this kernel and the three other quantized matrix-vector kernels, as the engine creates them (0112 and 0105 to 2 × 32, 0099 to 2 × 128, 0113 from 2 × 32 to 1 × 256): <strong>decoding went from 57 to 71 tokens per second, 1.20× as the median of the per-repetition pairings (1.18 to 1.33×) and 1.25× as the ratio of the condition medians (17.65 → 14.12 ms per token)</strong>, over five repetitions of each of six patch conditions, 30 runs, with no GPU errors. The per-repetition pairing is the aggregate the shuffled design supports: the condition order is reshuffled inside every repetition, and pairing within a repetition uses that blocking while the ratio of condition medians does not. <a href="https://huggingface.co/litert-community/gemma-4-E2B-it-litert-lm" target="_blank" rel="noopener noreferrer">Google's model card</a> reports 73 decode tokens per second for this same web bundle on a newer M4 Max, measured over 256 decode tokens after a 1,024-token prefill with a context length of 2,048 tokens, while the runs here time the last 67 to 69 tokens of a 76- to 78-token reply to a one-sentence prompt, so the two differ in context length as well as hardware and are not comparable (its 160.2 native figure is for a different, larger build). The generated text was identical apart from one word, which is consistent with a near-tie where a 32-way sum rounds differently from a 4-way one. In an earlier sweep (<code>out/patch.json</code>), 0112 patched alone left the text unchanged; in the 30-run sweep every patched set changes that one word, including the one that leaves 0105 unpatched. The in-model runs were made on this one GPU, all 30 in Chrome 146.</p>
|
| 196 |
+
<p>Four of those conditions leave one kernel unpatched, which measures each kernel's marginal contribution with the other three already patched: by the ratio of condition medians, without the 4-bit kernel 0099 it drops from 1.25× to 1.13×, and without the 4-bit 0105 to 1.15×, while without this page's kernel 0112 it is still 1.23× and without 0113 1.22×. <strong>So within the four-kernel patch the two 4-bit kernels make the largest marginal contributions, and 0112 and 0113 smaller ones.</strong> Those last two cannot be ordered against each other: dropping 0112 costs less than dropping 0113 under the ratio of condition medians and more under the per-repetition pairing, so which of them ranks last follows from the choice of aggregate. The 1.65× above is the isolated kernel's speedup in Chrome 146 (inferred from the adapter string; 0.84× on Chrome 131); over its 40 dispatches per token that predicts about 0.96 ms per token saved. Patched alone, in an earlier two-repetition sweep with a fixed condition order (<code>out/patch.json</code>, whose unpatched baseline of 18.4 to 18.9 ms makes it not directly comparable with the 30-run sweep), 0112 saved 0.690 and 0.885 ms per token, near that prediction. Leaving it out of the four-kernel patch costs 0.240 ms per token by the condition medians (per repetition −0.235 to 1.960 ms), so with the other three already patched its marginal contribution is smaller than its stand-alone saving in that earlier sweep; whether that reflects overlapping gains or the difference between the two sweeps is not established. 0099, a 4-bit kernel that also serves the 1.5 MiB q and o projections, moves the most in four of five repetitions.</p>
|
| 197 |
<p>This page's reference numbers are from an Apple M2 Max; the repository README holds the Colab T4 run. Copy yours with the button above.</p>
|
| 198 |
|
| 199 |
<footer>Independent work, not affiliated with Google. Kernel source emitted by <code>@litert-lm/core</code> 0.17.1, published by Google under the <a href="https://www.apache.org/licenses/LICENSE-2.0" target="_blank" rel="noopener noreferrer">Apache License 2.0</a>. Measurement code and text by the page's author. Method, harness and full record: <a href="https://github.com/abgnydn/litert-capture" target="_blank" rel="noopener noreferrer">github.com/abgnydn/litert-capture</a>.</footer>
|
|
|
|
| 205 |
const OUT_SLICES = 3072, IN_SLICES = 384
|
| 206 |
const WEIGHT_BYTES = OUT_SLICES * IN_SLICES * 4
|
| 207 |
const BATCH = 48, ROUNDS = 8
|
| 208 |
+
// 24 matrices = 108 MiB, well past the M2 Max's system-level cache; 16 (72 MiB) left a
|
| 209 |
// measurable share of 'cold' reads warm. Falls back to 8 on GPU out-of-memory,
|
| 210 |
// and the results panel then says so.
|
| 211 |
let NMAT = 24
|
|
|
|
| 218 |
{ key: 'split32', label: '32-way split', sub: '2 × 32 workgroup · 98,304 threads', ks: 32, wgX: 2 },
|
| 219 |
]
|
| 220 |
const HEADLINE = ['orig', 'split32']
|
| 221 |
+
// Apple M2 Max (30-core), 2026-09-22: mean of two interleaved harness runs' medians.
|
| 222 |
+
// Chrome 146 is inferred for these two records, which carry only the adapter string
|
| 223 |
+
// apple/metal-3; the provenanced out/microbench-0112-2026-09-24T07-36-19-655Z.json
|
| 224 |
+
// records Chrome/146.0.7680.153 for the same kernel on the same machine, at 1.638x.
|
| 225 |
+
const REFERENCE = {
|
| 226 |
orig: { cold: 60.9, hot: 44.8, err: 6.25e-3 }, wg256: { cold: 60.6, hot: 46.6, err: 6.25e-3 },
|
| 227 |
split16: { cold: 40.2, hot: 33.0, err: 3.44e-3 }, split32: { cold: 36.8, hot: 34.5, err: 2.04e-3 },
|
| 228 |
}
|
|
|
|
| 341 |
|
| 342 |
// ---------- small helpers ----------
|
| 343 |
const $ = (id) => document.getElementById(id)
|
| 344 |
+
const MINE_WHAT = $('mine-what').textContent
|
| 345 |
const gbs = (us) => WEIGHT_BYTES / (us * 1e-6) / 1e9
|
| 346 |
const median = (a) => { const s = [...a].sort((x, y) => x - y); return s[Math.floor(s.length / 2)] }
|
| 347 |
const f16 = (x) => { const f = new Float32Array([x]), u = new Uint32Array(f.buffer)[0]; const s = (u >> 16) & 0x8000, e = ((u >> 23) & 0xff) - 112, m = u & 0x7fffff; if (e <= 0) return s; if (e >= 31) return s | 0x7c00; let h = s | (e << 10) | (m >> 13); if ((m & 0x1fff) > 0x1000 || ((m & 0x1fff) === 0x1000 && (h & 1))) h++; return h }
|
|
|
|
| 381 |
const text = (x, y, s, attrs = {}) => { const t = add('text', { x, y, fill: 'var(--muted)', 'font-family': 'var(--mono)', 'font-size': 11, ...attrs }); t.textContent = s; return t }
|
| 382 |
add('rect', { x: 40, y: 22, width: 240, height: 96, rx: 3, fill: 'var(--orig-soft)', stroke: 'var(--orig)' })
|
| 383 |
text(160, 66, 'weights', { 'text-anchor': 'middle', fill: 'var(--ink)', 'font-family': 'var(--body)', 'font-size': 13, 'font-weight': 600 })
|
| 384 |
+
text(160, 84, '12,288 × 1,536 · 2-bit · 4.5 MiB', { 'text-anchor': 'middle' })
|
| 385 |
text(160, 14, '1,536 inputs wide', { 'text-anchor': 'middle' })
|
| 386 |
text(30, 70, '12,288 rows', { 'text-anchor': 'end' })
|
| 387 |
text(300, 74, '×', { fill: 'var(--ink)', 'font-size': 18 })
|
|
|
|
| 437 |
let gpuError = null
|
| 438 |
device.addEventListener('uncapturederror', (e) => { gpuError = String(e.error?.message ?? e) })
|
| 439 |
|
| 440 |
+
say(`Generating ${NMAT} random weight matrices (${(NMAT * WEIGHT_BYTES / 1048576).toFixed(0)} MiB)…`); await yieldUI()
|
| 441 |
const weights = []
|
| 442 |
for (let m = 0; m < NMAT; m++) {
|
| 443 |
const a = new Uint8Array(WEIGHT_BYTES)
|
|
|
|
| 552 |
// present
|
| 553 |
lastResult = { page: 'kernel-0112', version: 2, kernel: '@litert-lm/core 0.17.1 shader 0112', date: new Date().toISOString(), gpu: adapterInfo, userAgent: navigator.userAgent, timing: hasTs ? (quantized ? 'gpu-timestamps-quantized-100us' : 'gpu-timestamps') : 'cpu-batch', batch: B, rounds: ROUNDS, positions: 2, matrices: NMAT, weightBytes: WEIGHT_BYTES, variants: Object.fromEntries(VARIANTS.map((v) => [v.key, { ks: v.ks, wgX: v.wgX, threads: OUT_SLICES * v.ks }])), results }
|
| 554 |
const speed = results[HEADLINE[0]].cold / results[HEADLINE[1]].cold
|
| 555 |
+
const headlineOk = HEADLINE.every((k) => results[k].err < ERR_LIMIT)
|
| 556 |
+
$('mine-speedup').textContent = headlineOk ? `${speed.toFixed(2)}×` : '–'
|
| 557 |
+
$('mine-what').textContent = !headlineOk ? 'not shown: the original or the 32-way split gave a wrong answer on this GPU, so its speed does not count' : speed < 1 ? MINE_WHAT.replace(/^faster/, 'slower') : MINE_WHAT
|
| 558 |
$('mine-meta').textContent = `${[adapterInfo.vendor, adapterInfo.architecture].filter(Boolean).join(' · ') || 'this GPU'} · ${lastResult.timing.replace(/-/g, ' ')} · ${B * ROUNDS * 2} runs per number · ${NMAT} matrices${NMAT < 24 ? ' (reduced: GPU memory)' : ''}`
|
| 559 |
scaleMax = niceMax(Math.max(scaleMax, ...Object.values(results).map((d) => gbs(d.hot))))
|
| 560 |
renderRows($('mine-rows'), results, scaleMax); renderRows($('ref-rows'), REFERENCE, scaleMax)
|
|
|
|
| 565 |
device.destroy()
|
| 566 |
} catch (e) {
|
| 567 |
const msg = String(e?.message ?? e)
|
| 568 |
+
if (/out of memory|allocation|createBuffer|createTexture/i.test(msg) && NMAT > 8) { say(`Not enough GPU memory for ${NMAT} matrices; retrying with 8 (36 MiB). Cold reads will be partly cached.`); NMAT = 8; btn.disabled = false; return run() }
|
| 569 |
say('The measurement stopped: ' + msg)
|
| 570 |
}
|
| 571 |
btn.disabled = false
|
style.css
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
body {
|
| 2 |
-
padding: 2rem;
|
| 3 |
-
font-family: -apple-system, BlinkMacSystemFont, "Arial", sans-serif;
|
| 4 |
-
}
|
| 5 |
-
|
| 6 |
-
h1 {
|
| 7 |
-
font-size: 16px;
|
| 8 |
-
margin-top: 0;
|
| 9 |
-
}
|
| 10 |
-
|
| 11 |
-
p {
|
| 12 |
-
color: rgb(107, 114, 128);
|
| 13 |
-
font-size: 15px;
|
| 14 |
-
margin-bottom: 10px;
|
| 15 |
-
margin-top: 5px;
|
| 16 |
-
}
|
| 17 |
-
|
| 18 |
-
.card {
|
| 19 |
-
max-width: 620px;
|
| 20 |
-
margin: 0 auto;
|
| 21 |
-
padding: 16px;
|
| 22 |
-
border: 1px solid lightgray;
|
| 23 |
-
border-radius: 16px;
|
| 24 |
-
}
|
| 25 |
-
|
| 26 |
-
.card p:last-child {
|
| 27 |
-
margin-bottom: 0;
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|