9bow commited on
Commit
9412f7c
·
verified ·
1 Parent(s): f0047a1

Add qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2 WebTorch bundle

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/entrypoints/prefill-16.json +0 -0
  3. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/entrypoints/prefill-4.json +0 -0
  4. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/entrypoints/prefill-64.json +0 -0
  5. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/graph.json +3 -0
  6. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/004c06d76ce77979b63fe4450df54986355fd10cd2ef62adae86780e49e4a4a7.wgsl +10 -0
  7. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0112d93bb1aa347d8887ce1b95610a1a21fc8450ff8f59bcd37747fa6bf6fa52.wgsl +10 -0
  8. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0151a813c4a92adeb44f0e3b6108c77b29804118b588c9025805f724fce839b4.wgsl +24 -0
  9. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0337d8a473b6e485bd383533df7d373dbbfe7a5f3a982424004560b7213fbaa3.wgsl +11 -0
  10. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/04f26661cc9ac73bb7f39f66b76c4be31ee8774e094eeb41ed7d677e1b1d3fa9.wgsl +9 -0
  11. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/04f5cb70141e507d48bd44a76bfdf264a404a706e8a012502dce5a22ee75bab0.wgsl +9 -0
  12. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0554a443ecf6ead32fd2d58178b9ce3c592c89a09b3f63fa509ee947323bb497.wgsl +9 -0
  13. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0585cb3267bfc00a704de0f2e614880f64ddf97595ffee7d3b8f989dc1b6a190.wgsl +28 -0
  14. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/06eecde176beab9ff488ad533f2f4513e09bc49a30c4358f9648076f619f9992.wgsl +9 -0
  15. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0725908abf78c022f86deb338a06779f6f1346ebfe54ff5a43ee6b9eeb93c972.wgsl +9 -0
  16. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/07c7cf86b77afd3f62bb0629f608b6386614a3fb667b047c906f92569bb24410.wgsl +31 -0
  17. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/07cfabe2e2c2feff22a06c8296600e5ca21efb7444419f0a05e5ff6f48e0292d.wgsl +10 -0
  18. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0835fa80eaadb7c0e6d677e1fc622ad5dcedabb431e5d312f3b7aeb2252a7c7c.wgsl +8 -0
  19. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/08390b1553d11820d1d96b10add657b92acb8d26cd9f11acab8f16777ecfc10a.wgsl +21 -0
  20. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/09ea34c2533b929bd309e2bcbcf6d27fa02c731d82276ba6425db85961ea121b.wgsl +10 -0
  21. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0b11d3b2f5547a9e2b258c39e766ad14aefc19f8fe0be10e854af82088befd3e.wgsl +28 -0
  22. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0b8815b1efe176d55ab08530c0a9e1913cf6c9ef3e773c574843573d237cf196.wgsl +10 -0
  23. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0bbd6a5e356c1205a61dc3580a3ff68d88b98a952a5692d66b56ded171d9fc56.wgsl +9 -0
  24. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0d1edb63bf4dc212d55f8a4c79a50e04c0d822bd9f6e193003e2071ba0d57d69.wgsl +9 -0
  25. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0d3892883d1532d173a7dbe85d9ab1daa87f0d9b4923a535b8d2003f24f9500c.wgsl +11 -0
  26. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0e5468583a1ba929483119c50d4be42a58aed4dde1e15cc231908ed7543d0b71.wgsl +9 -0
  27. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0e9b6ee5f92769faaee5c746d1a3b936cdb7731f3f8e70cbf0a3fe96f6c8d921.wgsl +10 -0
  28. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0f466ed2664d77386f48afd19d8ab2734312bfd4f5f4d5b063d801a0b0a97cd9.wgsl +9 -0
  29. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0f943f386d2f37a8e56b4285b26e1c093c2f2fa2a5ce81cc43c44cc9bb0137f5.wgsl +9 -0
  30. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0fbc9484ef8ca703ced4ffd123f1ae733c44556b6cb160ee803b820ff5ec867d.wgsl +10 -0
  31. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/11a92e9cba6bfc381c98c5127db137639d6c87f1cb220459880ead039dc81256.wgsl +9 -0
  32. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/11dc9525598c8cb254a6722d441fa0d46fc51f383d426c38d5e3e2d8d3535b54.wgsl +31 -0
  33. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/121285fc94b4a9f873c60a1d6e0de9d764286607718405fd18288e7ff419c483.wgsl +11 -0
  34. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/12155d86d776e73f23d3bff0268c60dccb54c209ee9ee3a8cfd227ec4aa5b6df.wgsl +9 -0
  35. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/123232b7131301cfaa5ecccbaafc9f6c88fff84243969080a6fd194afd9c81d7.wgsl +17 -0
  36. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1255847146b31732939e573cdb26c483358c1ce07f987bd8c3d3d69ae8c7e141.wgsl +9 -0
  37. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/12cbaa151e326823e292ec43eff7eebe110ddd8c2c91b7a2a16e3fd5b04d0d44.wgsl +17 -0
  38. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/12cc6cd678e39876d00599e253b4da39a91e43a730fce08a96f163eb644ea094.wgsl +31 -0
  39. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/13237b557466d2fef5bfbc74e7ac77219b9821a53bc0843dbb5d14db8b04d414.wgsl +9 -0
  40. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1326034f3abab37629cd88f869d1b0bafcde724f85c07904b54feaaf71d89c62.wgsl +44 -0
  41. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/13dc73935b9b3b978232043e13babee3877bd53ca572be440de0590dacce8b17.wgsl +9 -0
  42. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/149c3fdb542967fc038109de3f2b7b2fbee1a6f594baffe28f6f16bafb84d355.wgsl +24 -0
  43. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1545025e0a625ccc73b42ae112c6f9d78a1504c10b759979ae2247984f25f056.wgsl +11 -0
  44. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/17ec6bdaf98edf7657b74a25036c2bec806ecbb4280f33fff6afd2e0de5ed377.wgsl +9 -0
  45. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/18a0b02a12ebf73203283d267342582d54ba411aeee9cbe21ad89ae890e4cfea.wgsl +9 -0
  46. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/18a66ec0875c456c9defd45d15e6ef0c148e64a1618e2825c7f96eadbbe8303a.wgsl +9 -0
  47. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/18b1cef2bae996f56664d9215180e3bfc77b4e514db52fdf9b6c58be51b53edb.wgsl +9 -0
  48. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/19564c145248f065669be23f5daeeda96e025ee4389bef80983001e83fddb31d.wgsl +11 -0
  49. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1a73b6fceda28f40bba4ae27a986ffa39842db94fa5903ea5bf689a7e854d4f3.wgsl +11 -0
  50. qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1ad205080cae29142a3ad87343a0cf3ac4945455398293946110d421ce6d515e.wgsl +10 -0
.gitattributes CHANGED
@@ -53,3 +53,4 @@ qwen35-08b-fp32-int2-g32-up-proj-mse-home-token-major-v2/graph.json filter=lfs d
53
  qwen35-08b-fp32-int2-g32-up-proj-mse-home-token-major-v2/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
54
  qwen35-2b-fp32-int8-g32-exclude-l6-l9-l10up-l15o-home-token-major-v2/graph.json filter=lfs diff=lfs merge=lfs -text
55
  qwen35-2b-fp32-int8-g32-exclude-l6-l9-l10up-l15o-home-token-major-v2/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
53
  qwen35-08b-fp32-int2-g32-up-proj-mse-home-token-major-v2/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
54
  qwen35-2b-fp32-int8-g32-exclude-l6-l9-l10up-l15o-home-token-major-v2/graph.json filter=lfs diff=lfs merge=lfs -text
55
  qwen35-2b-fp32-int8-g32-exclude-l6-l9-l10up-l15o-home-token-major-v2/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
56
+ qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/graph.json filter=lfs diff=lfs merge=lfs -text
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/entrypoints/prefill-16.json ADDED
The diff for this file is too large to render. See raw diff
 
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/entrypoints/prefill-4.json ADDED
The diff for this file is too large to render. See raw diff
 
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/entrypoints/prefill-64.json ADDED
The diff for this file is too large to render. See raw diff
 
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/graph.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d52601f3b10b29099e1e2919c53a00dc6b5dd37785f16333b64227c97855b07
3
+ size 10758848
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/004c06d76ce77979b63fe4450df54986355fd10cd2ef62adae86780e49e4a4a7.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ @compute @workgroup_size(64)
6
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
7
+ let i = gid.x + gid.y * 16384u;
8
+ if (i >= 16384u) { return; }
9
+ out[i] = f32(f32(b0[i]) + (f32(b1[i]) * 1.0));
10
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0112d93bb1aa347d8887ce1b95610a1a21fc8450ff8f59bcd37747fa6bf6fa52.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ @compute @workgroup_size(64)
6
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
7
+ let i = gid.x + gid.y * 2048u;
8
+ if (i >= 2048u) { return; }
9
+ out[i] = f32(f32(b0[i]) + (f32(b1[i]) * 1.0));
10
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0151a813c4a92adeb44f0e3b6108c77b29804118b588c9025805f724fce839b4.wgsl ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ var<workgroup> factor: f32;
6
+ @compute @workgroup_size(64)
7
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
8
+ let row = group.x + group.y * 128u;
9
+ if (row >= 128u) { return; }
10
+ let lane = local.x;
11
+ if (lane == 0u) {
12
+ var total = 0.0;
13
+ for (var j = 0u; j < 256u; j++) {
14
+ let v = f32(b0[row * 256u + j]);
15
+ total += v * v;
16
+ }
17
+ factor = inverseSqrt(total / 256.0 + 1e-06);
18
+ }
19
+ workgroupBarrier();
20
+ for (var p = lane; p < 256u; p += 64u) {
21
+ let i = row * 256u + p;
22
+ out[i] = f32(f32(b0[row * 256u + p]) * factor * (f32(b1[p]) + 1.0));
23
+ }
24
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0337d8a473b6e485bd383533df7d373dbbfe7a5f3a982424004560b7213fbaa3.wgsl ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 4096u;
7
+ if (i >= 4096u) { return; }
8
+ let coord = (i / 1u) % 64u;
9
+ if (coord >= 0u && coord < 32u) { out[i] = f32(b0[(i / 64u) * 32u + (coord - 0u) * 1u + i % 1u]); }
10
+ if (coord >= 32u && coord < 64u) { out[i] = f32(b0[(i / 64u) * 32u + (coord - 32u) * 1u + i % 1u]); }
11
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/04f26661cc9ac73bb7f39f66b76c4be31ee8774e094eeb41ed7d677e1b1d3fa9.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 64u;
7
+ if (i >= 10u) { return; }
8
+ out[i] = f32(b0[((i / 10u) % 1u) * 32u + ((i / 10u) % 1u) * 32u + (((i / 1u) % 10u) * 3u + 2u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/04f5cb70141e507d48bd44a76bfdf264a404a706e8a012502dce5a22ee75bab0.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 24576u;
7
+ if (i >= 24576u) { return; }
8
+ out[i] = f32(b0[((i / 24576u) % 1u) * 24576u + ((i / 4u) % 6144u) * 1u + ((i / 1u) % 4u) * 6144u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0554a443ecf6ead32fd2d58178b9ce3c592c89a09b3f63fa509ee947323bb497.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 128u;
7
+ if (i >= 128u) { return; }
8
+ out[i] = f32(b0[((i / 128u) % 1u) * 512u + ((i / 64u) % 2u) * 256u + ((i / 64u) % 1u) * 256u + (((i / 1u) % 64u) * 1u + 0u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0585cb3267bfc00a704de0f2e614880f64ddf97595ffee7d3b8f989dc1b6a190.wgsl ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<u32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+ fn unpack_bf16_1(index: u32) -> f32 {
5
+ let pair = b1[index / 2u];
6
+ let bits = (pair >> ((index % 2u) * 16u)) & 65535u;
7
+ return bitcast<f32>(bits << 16u);
8
+ }
9
+
10
+ var<workgroup> tile_a: array<f32, 64>;
11
+ var<workgroup> tile_b: array<f32, 64>;
12
+ @compute @workgroup_size(8, 8)
13
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
14
+ let row = group.y * 8u + local.y;
15
+ let col = group.x * 8u + local.x;
16
+ let i = (group.z * 64u + row) * 51712u + col;
17
+ var acc = 0.0;
18
+ for (var tile = 0u; tile < 1024u; tile += 8u) {
19
+ tile_a[local.y * 8u + local.x] = 0.0;
20
+ tile_b[local.x * 8u + local.y] = 0.0;
21
+ if (row < 64u && tile + local.x < 1024u) { tile_a[local.y * 8u + local.x] = f32(b0[(group.z * 64u + row) * 1024u + tile + local.x]); }
22
+ if (group.x * 8u + local.y < 51712u && tile + local.x < 1024u) { tile_b[local.x * 8u + local.y] = unpack_bf16_1((group.x * 8u + local.y) * 1024u + tile + local.x); }
23
+ workgroupBarrier();
24
+ for (var p = 0u; p < 8u; p++) { acc += tile_a[local.y * 8u + p] * tile_b[p * 8u + local.x]; }
25
+ workgroupBarrier();
26
+ }
27
+ if (row < 64u && col < 51712u) { out[i] = f32(acc + 0.0); }
28
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/06eecde176beab9ff488ad533f2f4513e09bc49a30c4358f9648076f619f9992.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 6144u;
7
+ if (i >= 6144u) { return; }
8
+ out[i] = f32(b0[((i / 2048u) % 3u) * 2048u + ((i / 2048u) % 1u) * 2048u + ((i / 64u) % 32u) * 1u + ((i / 1u) % 64u) * 32u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0725908abf78c022f86deb338a06779f6f1346ebfe54ff5a43ee6b9eeb93c972.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 64u;
7
+ if (i >= 44u) { return; }
8
+ out[i] = f32(b0[i]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/07c7cf86b77afd3f62bb0629f608b6386614a3fb667b047c906f92569bb24410.wgsl ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<u32>;
3
+ @group(0) @binding(2) var<storage, read> b2: array<f32>;
4
+ @group(0) @binding(3) var<storage, read_write> out: array<f32>;
5
+ fn dequant_1(index: u32) -> f32 {
6
+ let row = index / 1024u;
7
+ let col = index % 1024u;
8
+ let word = b1[row * 256u + col / 4u];
9
+ let code = i32((word >> ((col % 4u) * 8u)) & 255u) - 128;
10
+ return f32(code) * b2[row * 32u + col / 32u];
11
+ }
12
+
13
+ var<workgroup> tile_a: array<f32, 64>;
14
+ var<workgroup> tile_b: array<f32, 64>;
15
+ @compute @workgroup_size(8, 8)
16
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
17
+ let row = group.y * 8u + local.y;
18
+ let col = group.x * 8u + local.x;
19
+ let i = (group.z * 4u + row) * 4096u + col;
20
+ var acc = 0.0;
21
+ for (var tile = 0u; tile < 1024u; tile += 8u) {
22
+ tile_a[local.y * 8u + local.x] = 0.0;
23
+ tile_b[local.x * 8u + local.y] = 0.0;
24
+ if (row < 4u && tile + local.x < 1024u) { tile_a[local.y * 8u + local.x] = f32(b0[(group.z * 4u + row) * 1024u + tile + local.x]); }
25
+ if (group.x * 8u + local.y < 4096u && tile + local.x < 1024u) { tile_b[local.x * 8u + local.y] = dequant_1((group.x * 8u + local.y) * 1024u + tile + local.x); }
26
+ workgroupBarrier();
27
+ for (var p = 0u; p < 8u; p++) { acc += tile_a[local.y * 8u + p] * tile_b[p * 8u + local.x]; }
28
+ workgroupBarrier();
29
+ }
30
+ if (row < 4u && col < 4096u) { out[i] = f32(acc + 0.0); }
31
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/07cfabe2e2c2feff22a06c8296600e5ca21efb7444419f0a05e5ff6f48e0292d.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 1024u;
7
+ if (i >= 1024u) { return; }
8
+ let x = f32(b0[i]);
9
+ out[i] = f32(-x);
10
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0835fa80eaadb7c0e6d677e1fc622ad5dcedabb431e5d312f3b7aeb2252a7c7c.wgsl ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read_write> out: array<f32>;
2
+
3
+ @compute @workgroup_size(64)
4
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
5
+ let i = gid.x + gid.y * 2048u;
6
+ if (i >= 2048u) { return; }
7
+ out[i] = f32(0.0);
8
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/08390b1553d11820d1d96b10add657b92acb8d26cd9f11acab8f16777ecfc10a.wgsl ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ @compute @workgroup_size(64)
6
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
7
+ let i = gid.x + gid.y * 436224u;
8
+ if (i >= 436224u) { return; }
9
+ let batch = i / 436224u;
10
+ let channel = (i / 71u) % 6144u;
11
+ let group = channel / 1u;
12
+ var acc = 0.0;
13
+ for (var c = 0u; c < 1u; c++) {
14
+ for (var p = 0u; p < 4u; p++) {
15
+ let pos0 = i32((i / 1u) % 71u) * 1 + i32((p / 1u) % 4u) * 1 - 3;
16
+ if (pos0 >= 0 && pos0 < 68) { acc += f32(b0[batch * 417792u + (group * 1u + c) * 68u + u32(pos0) * 1u]) * f32(b1[(channel * 1u + c) * 4u + p]); }
17
+ }
18
+ }
19
+ out[i] = f32(acc + 0.0);
20
+
21
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/09ea34c2533b929bd309e2bcbcf6d27fa02c731d82276ba6425db85961ea121b.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 256u;
7
+ if (i >= 256u) { return; }
8
+ let x = f32(b0[i]);
9
+ out[i] = f32(inverseSqrt(x));
10
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0b11d3b2f5547a9e2b258c39e766ad14aefc19f8fe0be10e854af82088befd3e.wgsl ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<u32>;
3
+ @group(0) @binding(2) var<storage, read> b2: array<f32>;
4
+ @group(0) @binding(3) var<storage, read_write> out: array<f32>;
5
+ fn dequant_1(index: u32) -> f32 {
6
+ let row = index / 1024u;
7
+ let col = index % 1024u;
8
+ let word = b1[row * 256u + col / 4u];
9
+ let code = i32((word >> ((col % 4u) * 8u)) & 255u) - 128;
10
+ return f32(code) * b2[row * 32u + col / 32u];
11
+ }
12
+
13
+ var<workgroup> partial: array<f32, 64>;
14
+ @compute @workgroup_size(64)
15
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
16
+ let i = group.x + group.y * 6144u;
17
+ if (i >= 6144u) { return; }
18
+ let lane = local.x;
19
+ var acc = 0.0;
20
+ for (var p = lane; p < 1024u; p += 64u) { acc += f32(b0[(i / 6144u) * 1024u + p]) * dequant_1((i % 6144u) * 1024u + p); }
21
+ partial[lane] = acc;
22
+ workgroupBarrier();
23
+ for (var stride = 32u; stride > 0u; stride /= 2u) {
24
+ if (lane < stride) { partial[lane] += partial[lane + stride]; }
25
+ workgroupBarrier();
26
+ }
27
+ if (lane == 0u) { out[i] = f32(partial[0] + 0.0); }
28
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0b8815b1efe176d55ab08530c0a9e1913cf6c9ef3e773c574843573d237cf196.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 256u;
7
+ if (i >= 256u) { return; }
8
+ let x = f32(b0[i]);
9
+ out[i] = f32(sin(x));
10
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0bbd6a5e356c1205a61dc3580a3ff68d88b98a952a5692d66b56ded171d9fc56.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 1024u;
7
+ if (i >= 1024u) { return; }
8
+ out[i] = f32(f32(b0[i]) + (1e-06 * 1.0));
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0d1edb63bf4dc212d55f8a4c79a50e04c0d822bd9f6e193003e2071ba0d57d69.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 8192u;
7
+ if (i >= 8192u) { return; }
8
+ out[i] = f32(b0[((i / 8192u) % 1u) * 16384u + ((i / 2048u) % 4u) * 4096u + ((i / 256u) % 8u) * 512u + (((i / 1u) % 256u) * 1u + 256u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0d3892883d1532d173a7dbe85d9ab1daa87f0d9b4923a535b8d2003f24f9500c.wgsl ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 256u;
7
+ if (i >= 256u) { return; }
8
+ let coord = (i / 1u) % 64u;
9
+ if (coord >= 0u && coord < 32u) { out[i] = f32(b0[(i / 64u) * 32u + (coord - 0u) * 1u + i % 1u]); }
10
+ if (coord >= 32u && coord < 64u) { out[i] = f32(b0[(i / 64u) * 32u + (coord - 32u) * 1u + i % 1u]); }
11
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0e5468583a1ba929483119c50d4be42a58aed4dde1e15cc231908ed7543d0b71.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 512u;
7
+ if (i >= 512u) { return; }
8
+ out[i] = f32(b0[i]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0e9b6ee5f92769faaee5c746d1a3b936cdb7731f3f8e70cbf0a3fe96f6c8d921.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ @compute @workgroup_size(64)
6
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
7
+ let i = gid.x + gid.y * 32768u;
8
+ if (i >= 32768u) { return; }
9
+ out[i] = f32(f32(b0[i]) + (f32(b1[((i / 1u) % 4096u) * 1u]) * 1.0));
10
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0f466ed2664d77386f48afd19d8ab2734312bfd4f5f4d5b063d801a0b0a97cd9.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 2048u;
7
+ if (i >= 2048u) { return; }
8
+ out[i] = f32(b0[((i / 2048u) % 1u) * 2048u + ((i / 128u) % 16u) * 128u + ((i / 128u) % 1u) * 2048u + ((i / 1u) % 128u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0f943f386d2f37a8e56b4285b26e1c093c2f2fa2a5ce81cc43c44cc9bb0137f5.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 384u;
7
+ if (i >= 384u) { return; }
8
+ out[i] = f32(b0[((i / 128u) % 3u) * 128u + ((i / 128u) % 1u) * 128u + ((i / 32u) % 4u) * 1u + ((i / 1u) % 32u) * 4u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/0fbc9484ef8ca703ced4ffd123f1ae733c44556b6cb160ee803b820ff5ec867d.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ @compute @workgroup_size(64)
6
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
7
+ let i = gid.x + gid.y * 2048u;
8
+ if (i >= 2048u) { return; }
9
+ out[i] = f32(f32(b0[i]) * f32(b1[((i / 64u) % 16u) * 64u + ((i / 1u) % 64u) * 1u]));
10
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/11a92e9cba6bfc381c98c5127db137639d6c87f1cb220459880ead039dc81256.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<i32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<i32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 64u;
7
+ if (i >= 12u) { return; }
8
+ out[i] = i32(b0[(((i / 4u) % 3u) * 1u + 1u) * 4u + ((i / 4u) % 1u) * 4u + ((i / 1u) % 4u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/11dc9525598c8cb254a6722d441fa0d46fc51f383d426c38d5e3e2d8d3535b54.wgsl ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<u32>;
3
+ @group(0) @binding(2) var<storage, read> b2: array<f32>;
4
+ @group(0) @binding(3) var<storage, read_write> out: array<f32>;
5
+ fn dequant_1(index: u32) -> f32 {
6
+ let row = index / 1024u;
7
+ let col = index % 1024u;
8
+ let word = b1[row * 256u + col / 4u];
9
+ let code = i32((word >> ((col % 4u) * 8u)) & 255u) - 128;
10
+ return f32(code) * b2[row * 32u + col / 32u];
11
+ }
12
+
13
+ var<workgroup> tile_a: array<f32, 64>;
14
+ var<workgroup> tile_b: array<f32, 64>;
15
+ @compute @workgroup_size(8, 8)
16
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
17
+ let row = group.y * 8u + local.y;
18
+ let col = group.x * 8u + local.x;
19
+ let i = (group.z * 4u + row) * 6144u + col;
20
+ var acc = 0.0;
21
+ for (var tile = 0u; tile < 1024u; tile += 8u) {
22
+ tile_a[local.y * 8u + local.x] = 0.0;
23
+ tile_b[local.x * 8u + local.y] = 0.0;
24
+ if (row < 4u && tile + local.x < 1024u) { tile_a[local.y * 8u + local.x] = f32(b0[(group.z * 4u + row) * 1024u + tile + local.x]); }
25
+ if (group.x * 8u + local.y < 6144u && tile + local.x < 1024u) { tile_b[local.x * 8u + local.y] = dequant_1((group.x * 8u + local.y) * 1024u + tile + local.x); }
26
+ workgroupBarrier();
27
+ for (var p = 0u; p < 8u; p++) { acc += tile_a[local.y * 8u + p] * tile_b[p * 8u + local.x]; }
28
+ workgroupBarrier();
29
+ }
30
+ if (row < 4u && col < 6144u) { out[i] = f32(acc + 0.0); }
31
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/121285fc94b4a9f873c60a1d6e0de9d764286607718405fd18288e7ff419c483.wgsl ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 256u;
7
+ if (i >= 256u) { return; }
8
+ let x = f32(b0[i]);
9
+ if (x * 1.0 > 20.0) { out[i] = f32(x); return; }
10
+ out[i] = f32(log(1.0 + exp(x * 1.0)) / 1.0);
11
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/12155d86d776e73f23d3bff0268c60dccb54c209ee9ee3a8cfd227ec4aa5b6df.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 1024u;
7
+ if (i >= 1024u) { return; }
8
+ out[i] = f32(b0[i]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/123232b7131301cfaa5ecccbaafc9f6c88fff84243969080a6fd194afd9c81d7.wgsl ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<i32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<u32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+ fn unpack_bf16_1(index: u32) -> f32 {
5
+ let pair = b1[index / 2u];
6
+ let bits = (pair >> ((index % 2u) * 16u)) & 65535u;
7
+ return bitcast<f32>(bits << 16u);
8
+ }
9
+
10
+ @compute @workgroup_size(64)
11
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
12
+ let i = gid.x + gid.y * 4096u;
13
+ if (i >= 4096u) { return; }
14
+ let token = i32(b0[i / 1024u]);
15
+ if (token < 196608 || token >= 248320) { out[i] = f32(0.0); return; }
16
+ out[i] = f32(unpack_bf16_1(u32(token - 196608) * 1024u + i % 1024u));
17
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1255847146b31732939e573cdb26c483358c1ce07f987bd8c3d3d69ae8c7e141.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 24576u;
7
+ if (i >= 24576u) { return; }
8
+ out[i] = f32(b0[((i / 24576u) % 1u) * 30720u + ((i / 4u) % 6144u) * 5u + (((i / 1u) % 4u) * 1u + 1u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/12cbaa151e326823e292ec43eff7eebe110ddd8c2c91b7a2a16e3fd5b04d0d44.wgsl ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<i32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<u32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+ fn unpack_bf16_1(index: u32) -> f32 {
5
+ let pair = b1[index / 2u];
6
+ let bits = (pair >> ((index % 2u) * 16u)) & 65535u;
7
+ return bitcast<f32>(bits << 16u);
8
+ }
9
+
10
+ @compute @workgroup_size(64)
11
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
12
+ let i = gid.x + gid.y * 16384u;
13
+ if (i >= 16384u) { return; }
14
+ let token = i32(b0[i / 1024u]);
15
+ if (token < 65536 || token >= 131072) { out[i] = f32(0.0); return; }
16
+ out[i] = f32(unpack_bf16_1(u32(token - 65536) * 1024u + i % 1024u));
17
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/12cc6cd678e39876d00599e253b4da39a91e43a730fce08a96f163eb644ea094.wgsl ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<u32>;
3
+ @group(0) @binding(2) var<storage, read> b2: array<f32>;
4
+ @group(0) @binding(3) var<storage, read_write> out: array<f32>;
5
+ fn dequant_1(index: u32) -> f32 {
6
+ let row = index / 1024u;
7
+ let col = index % 1024u;
8
+ let word = b1[row * 256u + col / 4u];
9
+ let code = i32((word >> ((col % 4u) * 8u)) & 255u) - 128;
10
+ return f32(code) * b2[row * 32u + col / 32u];
11
+ }
12
+
13
+ var<workgroup> tile_a: array<f32, 64>;
14
+ var<workgroup> tile_b: array<f32, 64>;
15
+ @compute @workgroup_size(8, 8)
16
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
17
+ let row = group.y * 8u + local.y;
18
+ let col = group.x * 8u + local.x;
19
+ let i = (group.z * 4u + row) * 2048u + col;
20
+ var acc = 0.0;
21
+ for (var tile = 0u; tile < 1024u; tile += 8u) {
22
+ tile_a[local.y * 8u + local.x] = 0.0;
23
+ tile_b[local.x * 8u + local.y] = 0.0;
24
+ if (row < 4u && tile + local.x < 1024u) { tile_a[local.y * 8u + local.x] = f32(b0[(group.z * 4u + row) * 1024u + tile + local.x]); }
25
+ if (group.x * 8u + local.y < 2048u && tile + local.x < 1024u) { tile_b[local.x * 8u + local.y] = dequant_1((group.x * 8u + local.y) * 1024u + tile + local.x); }
26
+ workgroupBarrier();
27
+ for (var p = 0u; p < 8u; p++) { acc += tile_a[local.y * 8u + p] * tile_b[p * 8u + local.x]; }
28
+ workgroupBarrier();
29
+ }
30
+ if (row < 4u && col < 2048u) { out[i] = f32(acc + 0.0); }
31
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/13237b557466d2fef5bfbc74e7ac77219b9821a53bc0843dbb5d14db8b04d414.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 2048u;
7
+ if (i >= 2048u) { return; }
8
+ out[i] = f32(f32(b0[i]) * f32(b0[i]));
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1326034f3abab37629cd88f869d1b0bafcde724f85c07904b54feaaf71d89c62.wgsl ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read> b2: array<f32>;
4
+ @group(0) @binding(3) var<storage, read> b3: array<f32>;
5
+ @group(0) @binding(4) var<storage, read> b4: array<f32>;
6
+ @group(0) @binding(5) var<storage, read> b5: array<f32>;
7
+ @group(0) @binding(6) var<storage, read_write> out: array<f32>;
8
+
9
+ var<workgroup> partial: array<f32, 128>;
10
+ var<workgroup> delta: f32;
11
+ @compute @workgroup_size(128)
12
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
13
+ let column = group.x;
14
+ let head = group.y;
15
+ let batch_index = group.z;
16
+ let lane = local.x;
17
+ var recurrent = 0.0;
18
+ if (lane < 128u) { recurrent = f32(b5[((batch_index * 16u + head) * 128u + lane) * 128u + column]); }
19
+ for (var token = 0u; token < 4u; token++) {
20
+ recurrent *= exp(f32(b3[(batch_index * 4u + token) * 16u + head]));
21
+ partial[lane] = 0.0;
22
+ if (lane < 128u) { partial[lane] = recurrent * f32(b1[((batch_index * 4u + token) * 16u + head) * 128u + lane]); }
23
+ workgroupBarrier();
24
+ for (var stride = 64u; stride > 0u; stride /= 2u) {
25
+ if (lane < stride) { partial[lane] += partial[lane + stride]; }
26
+ workgroupBarrier();
27
+ }
28
+ if (lane == 0u) { delta = (f32(b2[((batch_index * 4u + token) * 16u + head) * 128u + column]) - partial[0]) * f32(b4[(batch_index * 4u + token) * 16u + head]); }
29
+ workgroupBarrier();
30
+ if (lane < 128u) { recurrent += f32(b1[((batch_index * 4u + token) * 16u + head) * 128u + lane]) * delta; }
31
+ partial[lane] = 0.0;
32
+ if (lane < 128u) {
33
+ partial[lane] = recurrent * f32(b0[((batch_index * 4u + token) * 16u + head) * 128u + lane]) * 0.08838834764831845;
34
+ }
35
+ workgroupBarrier();
36
+ for (var stride = 64u; stride > 0u; stride /= 2u) {
37
+ if (lane < stride) { partial[lane] += partial[lane + stride]; }
38
+ workgroupBarrier();
39
+ }
40
+ if (lane == 0u) { out[((batch_index * 16u + head) * 132u + 128u + token) * 128u + column] = f32(partial[0]); }
41
+ workgroupBarrier();
42
+ }
43
+ if (lane < 128u) { out[((batch_index * 16u + head) * 132u + lane) * 128u + column] = f32(recurrent); }
44
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/13dc73935b9b3b978232043e13babee3877bd53ca572be440de0590dacce8b17.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 64u;
7
+ if (i >= 32u) { return; }
8
+ out[i] = f32(b0[0u * 32u + ((i / 32u) % 1u) * 32u + ((i / 32u) % 1u) * 32u + ((i / 1u) % 32u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/149c3fdb542967fc038109de3f2b7b2fbee1a6f594baffe28f6f16bafb84d355.wgsl ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ var<workgroup> factor: f32;
6
+ @compute @workgroup_size(64)
7
+ fn main(@builtin(workgroup_id) group: vec3<u32>, @builtin(local_invocation_id) local: vec3<u32>) {
8
+ let row = group.x + group.y * 4u;
9
+ if (row >= 4u) { return; }
10
+ let lane = local.x;
11
+ if (lane == 0u) {
12
+ var total = 0.0;
13
+ for (var j = 0u; j < 1024u; j++) {
14
+ let v = f32(b0[row * 1024u + j]);
15
+ total += v * v;
16
+ }
17
+ factor = inverseSqrt(total / 1024.0 + 1e-06);
18
+ }
19
+ workgroupBarrier();
20
+ for (var p = lane; p < 1024u; p += 64u) {
21
+ let i = row * 1024u + p;
22
+ out[i] = f32(f32(b0[row * 1024u + p]) * factor * (f32(b1[p]) + 1.0));
23
+ }
24
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1545025e0a625ccc73b42ae112c6f9d78a1504c10b759979ae2247984f25f056.wgsl ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ @compute @workgroup_size(64)
6
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
7
+ let i = gid.x + gid.y * 128u;
8
+ if (i >= 128u) { return; }
9
+ let coord = (i / 1u) % 32u;
10
+ if (coord >= 1u && coord < 34u && (coord - 1u) % 3u == 0u) { out[i] = f32(b0[((i / 128u) % 1u) * 44u + ((i / 32u) % 4u) * 11u + ((((i / 1u) % 32u) - 1u) / 3u) * 1u]); } else { out[i] = f32(b1[i]); }
11
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/17ec6bdaf98edf7657b74a25036c2bec806ecbb4280f33fff6afd2e0de5ed377.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 2048u;
7
+ if (i >= 2048u) { return; }
8
+ out[i] = f32(f32(b0[i]) * 0.08838834764831843);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/18a0b02a12ebf73203283d267342582d54ba411aeee9cbe21ad89ae890e4cfea.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 8192u;
7
+ if (i >= 8192u) { return; }
8
+ out[i] = f32(b0[((i / 8192u) % 1u) * 8192u + ((i / 1024u) % 8u) * 256u + ((i / 256u) % 4u) * 2048u + ((i / 1u) % 256u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/18a66ec0875c456c9defd45d15e6ef0c148e64a1618e2825c7f96eadbbe8303a.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<i32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<i32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 64u;
7
+ if (i >= 16u) { return; }
8
+ out[i] = i32(b0[i]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/18b1cef2bae996f56664d9215180e3bfc77b4e514db52fdf9b6c58be51b53edb.wgsl ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 64u;
7
+ if (i >= 44u) { return; }
8
+ out[i] = f32(b0[((i / 44u) % 1u) * 128u + ((i / 11u) % 4u) * 32u + (((i / 1u) % 11u) * 3u + 1u) * 1u]);
9
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/19564c145248f065669be23f5daeeda96e025ee4389bef80983001e83fddb31d.wgsl ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 64u;
7
+ if (i >= 64u) { return; }
8
+ let x = f32(b0[i]);
9
+ if (x * 1.0 > 20.0) { out[i] = f32(x); return; }
10
+ out[i] = f32(log(1.0 + exp(x * 1.0)) / 1.0);
11
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1a73b6fceda28f40bba4ae27a986ffa39842db94fa5903ea5bf689a7e854d4f3.wgsl ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read_write> out: array<f32>;
3
+
4
+ @compute @workgroup_size(64)
5
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
6
+ let i = gid.x + gid.y * 1024u;
7
+ if (i >= 1024u) { return; }
8
+ let coord = (i / 1u) % 64u;
9
+ if (coord >= 0u && coord < 32u) { out[i] = f32(b0[(i / 64u) * 32u + (coord - 0u) * 1u + i % 1u]); }
10
+ if (coord >= 32u && coord < 64u) { out[i] = f32(b0[(i / 64u) * 32u + (coord - 32u) * 1u + i % 1u]); }
11
+ }
qwen35-08b-fp32-int8-attn-int4-mlp-gptq-home-token-major-v2/kernels/1ad205080cae29142a3ad87343a0cf3ac4945455398293946110d421ce6d515e.wgsl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ @group(0) @binding(0) var<storage, read> b0: array<f32>;
2
+ @group(0) @binding(1) var<storage, read> b1: array<f32>;
3
+ @group(0) @binding(2) var<storage, read_write> out: array<f32>;
4
+
5
+ @compute @workgroup_size(64)
6
+ fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
7
+ let i = gid.x + gid.y * 128u;
8
+ if (i >= 128u) { return; }
9
+ out[i] = f32(f32(b0[i]) + (f32(b1[i]) * 1.0));
10
+ }