Xenova HF Staff commited on
Commit
163b53c
·
verified ·
1 Parent(s): f430939

sync 6fdf6301e2bb

Browse files
README.md CHANGED
@@ -52,7 +52,7 @@ One implementation is selected per call from the device capabilities, the reques
52
  - [`metadata.json`](build/webgpu/metadata.json) — kernel metadata (id, digests, per-variant templates, provenance)
53
  - [`manifest.json`](build/webgpu/manifest.json) — the op contract (source of truth)
54
  - [`test.json`](build/webgpu/test.json) — correctness cases
55
- - [`bench.json`](build/webgpu/bench.json) — benchmark + tuning cases
56
  - [`dynamic-quantize-linear-quantize.wgsl.jinja`](build/webgpu/dynamic-quantize-linear-quantize.wgsl.jinja)
57
  - [`dynamic-quantize-linear-reduce.wgsl.jinja`](build/webgpu/dynamic-quantize-linear-reduce.wgsl.jinja)
58
  - [`dynamic-quantize-linear.wgsl.jinja`](build/webgpu/dynamic-quantize-linear.wgsl.jinja)
@@ -60,7 +60,7 @@ One implementation is selected per call from the device capabilities, the reques
60
  ## Use with `@huggingface/kernels`
61
 
62
  ```sh
63
- npm install --save-exact @huggingface/kernels@0.0.1-preview.2
64
  ```
65
 
66
  Required output shapes and logical data types are inferred from the supplied inputs and attributes; result tensors are allocated automatically.
 
52
  - [`metadata.json`](build/webgpu/metadata.json) — kernel metadata (id, digests, per-variant templates, provenance)
53
  - [`manifest.json`](build/webgpu/manifest.json) — the op contract (source of truth)
54
  - [`test.json`](build/webgpu/test.json) — correctness cases
55
+ - [`bench.json`](build/webgpu/bench.json) — benchmark cases
56
  - [`dynamic-quantize-linear-quantize.wgsl.jinja`](build/webgpu/dynamic-quantize-linear-quantize.wgsl.jinja)
57
  - [`dynamic-quantize-linear-reduce.wgsl.jinja`](build/webgpu/dynamic-quantize-linear-reduce.wgsl.jinja)
58
  - [`dynamic-quantize-linear.wgsl.jinja`](build/webgpu/dynamic-quantize-linear.wgsl.jinja)
 
60
  ## Use with `@huggingface/kernels`
61
 
62
  ```sh
63
+ npm install --save-exact @huggingface/kernels@0.0.1-preview.3
64
  ```
65
 
66
  Required output shapes and logical data types are inferred from the supplied inputs and attributes; result tensors are allocated automatically.
build/webgpu/bench.json CHANGED
@@ -69,7 +69,7 @@
69
  }
70
  },
71
  {
72
- "name": "alignment_healthy_vec4_4194304",
73
  "preset": "smoke",
74
  "inputs": {
75
  "x": {
 
69
  }
70
  },
71
  {
72
+ "name": "alignment_control_vec4_4194304",
73
  "preset": "smoke",
74
  "inputs": {
75
  "x": {
build/webgpu/dynamic-quantize-linear-quantize.wgsl.jinja CHANGED
@@ -140,7 +140,6 @@ fn round_dynamic_half_to_even(value: f32, scale: f32) -> i32 {
140
  return i32(select(upper, lower, lower_is_even));
141
  }
142
 
143
-
144
  const WG: u32 = {{ workgroupSize }}u;
145
  {% if not vec4 %}
146
  const EPT: u32 = {{ elemsPerThread }}u;
 
140
  return i32(select(upper, lower, lower_is_even));
141
  }
142
 
 
143
  const WG: u32 = {{ workgroupSize }}u;
144
  {% if not vec4 %}
145
  const EPT: u32 = {{ elemsPerThread }}u;
build/webgpu/dynamic-quantize-linear-reduce.wgsl.jinja CHANGED
@@ -27,13 +27,15 @@ var<workgroup> wgMax: array<f32, WG>;
27
 
28
  @compute @workgroup_size(WG, 1, 1)
29
  fn main(@builtin(workgroup_id) wg: vec3<u32>,
30
- @builtin(local_invocation_id) lid: vec3<u32>
31
- {%- if gridStride %},
32
- @builtin(num_workgroups) nwg: vec3<u32>
33
- {%- endif %}
34
- {%- if useSubgroups %},
 
35
  @builtin(subgroup_size) sgSize: u32
36
- {%- endif %}) {
 
37
  let tid = lid.x;
38
  {% if gridStride %}
39
  let blk = wg.x;
@@ -98,24 +100,29 @@ fn main(@builtin(workgroup_id) wg: vec3<u32>,
98
  let sgMin = subgroupMin(localMin);
99
  let sgMax = subgroupMax(localMax);
100
  // Cross-subgroup fold that assumes nothing about which invocations share a
101
- // subgroup, how many subgroups there are, or which of a subgroup's lanes are
102
- // active: every invocation owns the slot at its own index, the elected lane
 
103
  // publishes its subgroup pair there and every other lane publishes 0.0, which
104
  // is an exact identity here because every lane's local range already includes
105
  // zero (so every published minimum is <= 0 and every maximum >= 0). Each
106
- // subgroup then folds all WG slots — lane `rank`, its dense position among
107
- // the active lanes, walks slots rank, rank + count, ... — and one more
108
- // collective merges the lane partials, so every slot is merged exactly once
109
- // at any legal width and partition. min/max is commutative and associative,
110
- // so the fold order does not change the result.
111
  var totalMin = sgMin;
112
  var totalMax = sgMax;
113
  // A one-subgroup workgroup is already fully reduced by the collectives above.
114
  // The test reads the `subgroup_size` builtin, which is uniform; a collective's
115
  // result is not uniform to WGSL's analysis and may not guard a barrier.
116
  if (sgSize != WG) {
117
- let rank = subgroupExclusiveAdd(1u);
118
- let count = subgroupAdd(1u);
 
 
 
 
 
119
  let leader = rank == 0u;
120
  wgMin[tid] = select(0.0, sgMin, leader);
121
  wgMax[tid] = select(0.0, sgMax, leader);
 
27
 
28
  @compute @workgroup_size(WG, 1, 1)
29
  fn main(@builtin(workgroup_id) wg: vec3<u32>,
30
+ @builtin(local_invocation_id) lid: vec3<u32>{{ "," if gridStride or useSubgroups else "" }}
31
+ {% if gridStride %}
32
+ @builtin(num_workgroups) nwg: vec3<u32>{{ "," if useSubgroups else "" }}
33
+ {% endif %}
34
+ {% if useSubgroups %}
35
+ @builtin(subgroup_invocation_id) sgLane: u32,
36
  @builtin(subgroup_size) sgSize: u32
37
+ {% endif %}
38
+ ) {
39
  let tid = lid.x;
40
  {% if gridStride %}
41
  let blk = wg.x;
 
100
  let sgMin = subgroupMin(localMin);
101
  let sgMax = subgroupMax(localMax);
102
  // Cross-subgroup fold that assumes nothing about which invocations share a
103
+ // subgroup or how many subgroups there are (it does require the uniform
104
+ // control flow this entry point already has): every invocation owns the slot
105
+ // at its own index, the elected lane
106
  // publishes its subgroup pair there and every other lane publishes 0.0, which
107
  // is an exact identity here because every lane's local range already includes
108
  // zero (so every published minimum is <= 0 and every maximum >= 0). Each
109
+ // subgroup then folds all WG slots — lane `sgLane` walks slots sgLane,
110
+ // sgLane + sgSize, ... — and one more collective merges the lane partials, so
111
+ // every slot is merged exactly once at any legal width. min/max is commutative
112
+ // and associative, so the fold order does not change the result.
 
113
  var totalMin = sgMin;
114
  var totalMax = sgMax;
115
  // A one-subgroup workgroup is already fully reduced by the collectives above.
116
  // The test reads the `subgroup_size` builtin, which is uniform; a collective's
117
  // result is not uniform to WGSL's analysis and may not guard a barrier.
118
  if (sgSize != WG) {
119
+ // The coordinates are the BUILTINS, never `subgroupExclusiveAdd(1u)` /
120
+ // `subgroupAdd(1u)`: those agree with them under this contract, but a driver
121
+ // in the wild answers a claim derived from them with zero and leaves most of
122
+ // the workgroup's slots unclaimed. Every lane of this entry point is active
123
+ // here, so the full subgroup width is the right stride.
124
+ let rank = sgLane;
125
+ let count = sgSize;
126
  let leader = rank == 0u;
127
  wgMin[tid] = select(0.0, sgMin, leader);
128
  wgMax[tid] = select(0.0, sgMax, leader);
build/webgpu/dynamic-quantize-linear.wgsl.jinja CHANGED
@@ -142,7 +142,6 @@ fn round_dynamic_half_to_even(value: f32, scale: f32) -> i32 {
142
  return i32(select(upper, lower, lower_is_even));
143
  }
144
 
145
-
146
  @compute @workgroup_size(1)
147
  fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
148
  if (gid.x != 0u) { return; }
 
142
  return i32(select(upper, lower, lower_is_even));
143
  }
144
 
 
145
  @compute @workgroup_size(1)
146
  fn main(@builtin(global_invocation_id) gid: vec3<u32>) {
147
  if (gid.x != 0u) { return; }
build/webgpu/manifest.json CHANGED
@@ -35,17 +35,17 @@
35
  "serialFallbackNeeded": "inputCount <= tunables.SERIAL_MAX_ELEMENTS or not parallelFullFits"
36
  },
37
  "bindings": {
38
- "y": { "buffer": "storage", "elementType": "u32" },
39
- "y_scale": { "buffer": "storage", "elementType": "f32", "length": 1 },
40
- "y_zero_point": { "buffer": "storage", "elementType": "u32", "length": 1 },
41
- "params": { "buffer": "uniform", "struct": [{ "name": "count", "type": "u32", "value": "numel(shapes.x)" }] },
42
- "x_2": { "name": "x", "buffer": "read-only-storage", "elementType": "$inputElement" },
43
  "partial_min": { "buffer": "storage", "elementType": "f32" },
44
  "partial_max": { "buffer": "storage", "elementType": "f32" },
45
- "partial_min_2": { "name": "partial_min", "buffer": "read-only-storage", "elementType": "f32" },
46
- "partial_max_2": { "name": "partial_max", "buffer": "read-only-storage", "elementType": "f32" },
47
- "y_scale_2": { "name": "y_scale", "buffer": "read-only-storage", "elementType": "f32", "length": 1 },
48
- "y_zero_point_2": { "name": "y_zero_point", "buffer": "read-only-storage", "elementType": "u32", "length": 1 }
49
  },
50
  "variants": [
51
  {
@@ -82,8 +82,7 @@
82
  "id": "reduce",
83
  "name": "DynamicQuantizeLinear.ReduceMinMax",
84
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
85
- "subgroupCollectivesWidth": "portable",
86
- "bindings": ["x_2", "partial_min", "partial_max", "params"],
87
  "dispatch": { "x": "min(fullPartials, 65535)", "y": "ceilDiv(fullPartials, 65535)", "z": 1 }
88
  },
89
  {
@@ -92,8 +91,8 @@
92
  "shader": "dynamic-quantize-linear.wgsl.jinja",
93
  "derive": { "fromPartials": true },
94
  "bindings": [
95
- "partial_min_2",
96
- "partial_max_2",
97
  "y_scale",
98
  "y_zero_point",
99
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "fullPartials" }] }
@@ -104,10 +103,10 @@
104
  "id": "quantize",
105
  "name": "DynamicQuantizeLinear.Quantize",
106
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
107
- "bindings": ["x_2", "y_scale_2", "y_zero_point_2", "y", "params"],
108
  "dispatch": {
109
- "x": "min(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
110
- "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
111
  "z": 1
112
  }
113
  }
@@ -133,8 +132,7 @@
133
  "id": "reduce",
134
  "name": "DynamicQuantizeLinear.ReduceMinMax",
135
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
136
- "subgroupCollectivesWidth": "portable",
137
- "bindings": ["x_2", "partial_min", "partial_max", "params"],
138
  "dispatch": { "x": "min(fullPartials, 65535)", "y": "ceilDiv(fullPartials, 65535)", "z": 1 }
139
  },
140
  {
@@ -143,8 +141,8 @@
143
  "shader": "dynamic-quantize-linear.wgsl.jinja",
144
  "derive": { "fromPartials": true },
145
  "bindings": [
146
- "partial_min_2",
147
- "partial_max_2",
148
  "y_scale",
149
  "y_zero_point",
150
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "fullPartials" }] }
@@ -155,10 +153,10 @@
155
  "id": "quantize",
156
  "name": "DynamicQuantizeLinear.Quantize",
157
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
158
- "bindings": ["x_2", "y_scale_2", "y_zero_point_2", "y", "params"],
159
  "dispatch": {
160
- "x": "min(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
161
- "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
162
  "z": 1
163
  }
164
  }
@@ -184,9 +182,8 @@
184
  "id": "reduce",
185
  "name": "DynamicQuantizeLinear.ReduceMinMax",
186
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
187
- "subgroupCollectivesWidth": "portable",
188
  "derive": { "gridStride": true },
189
- "bindings": ["x_2", "partial_min", "partial_max", "params"],
190
  "dispatch": { "x": "gridPartials" }
191
  },
192
  {
@@ -195,8 +192,8 @@
195
  "shader": "dynamic-quantize-linear.wgsl.jinja",
196
  "derive": { "fromPartials": true },
197
  "bindings": [
198
- "partial_min_2",
199
- "partial_max_2",
200
  "y_scale",
201
  "y_zero_point",
202
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "gridPartials" }] }
@@ -207,10 +204,10 @@
207
  "id": "quantize",
208
  "name": "DynamicQuantizeLinear.Quantize",
209
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
210
- "bindings": ["x_2", "y_scale_2", "y_zero_point_2", "y", "params"],
211
  "dispatch": {
212
- "x": "min(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
213
- "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
214
  "z": 1
215
  }
216
  }
@@ -236,9 +233,8 @@
236
  "id": "reduce",
237
  "name": "DynamicQuantizeLinear.ReduceMinMax",
238
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
239
- "subgroupCollectivesWidth": "portable",
240
  "derive": { "gridStride": true },
241
- "bindings": ["x_2", "partial_min", "partial_max", "params"],
242
  "dispatch": { "x": "gridPartials" }
243
  },
244
  {
@@ -247,8 +243,8 @@
247
  "shader": "dynamic-quantize-linear.wgsl.jinja",
248
  "derive": { "fromPartials": true },
249
  "bindings": [
250
- "partial_min_2",
251
- "partial_max_2",
252
  "y_scale",
253
  "y_zero_point",
254
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "gridPartials" }] }
@@ -259,10 +255,10 @@
259
  "id": "quantize",
260
  "name": "DynamicQuantizeLinear.Quantize",
261
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
262
- "bindings": ["x_2", "y_scale_2", "y_zero_point_2", "y", "params"],
263
  "dispatch": {
264
- "x": "min(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
265
- "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
266
  "z": 1
267
  }
268
  }
 
35
  "serialFallbackNeeded": "inputCount <= tunables.SERIAL_MAX_ELEMENTS or not parallelFullFits"
36
  },
37
  "bindings": {
38
+ "y": { "elementType": "u32" },
39
+ "y_scale": { "elementType": "f32", "length": 1 },
40
+ "y_zero_point": { "elementType": "u32", "length": 1 },
41
+ "params": { "struct": [{ "name": "count", "type": "u32", "value": "numel(shapes.x)" }] },
42
+ "x_reduce": { "name": "x", "elementType": "$inputElement" },
43
  "partial_min": { "buffer": "storage", "elementType": "f32" },
44
  "partial_max": { "buffer": "storage", "elementType": "f32" },
45
+ "partial_min_f32": { "name": "partial_min", "buffer": "read-only-storage", "elementType": "f32" },
46
+ "partial_max_f32": { "name": "partial_max", "buffer": "read-only-storage", "elementType": "f32" },
47
+ "y_scale_f32": { "name": "y_scale", "buffer": "read-only-storage", "elementType": "f32", "length": 1 },
48
+ "y_zero_point_u32": { "name": "y_zero_point", "buffer": "read-only-storage", "elementType": "u32", "length": 1 }
49
  },
50
  "variants": [
51
  {
 
82
  "id": "reduce",
83
  "name": "DynamicQuantizeLinear.ReduceMinMax",
84
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
85
+ "bindings": ["x_reduce", "partial_min", "partial_max", "params"],
 
86
  "dispatch": { "x": "min(fullPartials, 65535)", "y": "ceilDiv(fullPartials, 65535)", "z": 1 }
87
  },
88
  {
 
91
  "shader": "dynamic-quantize-linear.wgsl.jinja",
92
  "derive": { "fromPartials": true },
93
  "bindings": [
94
+ "partial_min_f32",
95
+ "partial_max_f32",
96
  "y_scale",
97
  "y_zero_point",
98
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "fullPartials" }] }
 
103
  "id": "quantize",
104
  "name": "DynamicQuantizeLinear.Quantize",
105
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
106
+ "bindings": ["x_reduce", "y_scale_f32", "y_zero_point_u32", "y", "params"],
107
  "dispatch": {
108
+ "x": "min(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
109
+ "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
110
  "z": 1
111
  }
112
  }
 
132
  "id": "reduce",
133
  "name": "DynamicQuantizeLinear.ReduceMinMax",
134
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
135
+ "bindings": ["x_reduce", "partial_min", "partial_max", "params"],
 
136
  "dispatch": { "x": "min(fullPartials, 65535)", "y": "ceilDiv(fullPartials, 65535)", "z": 1 }
137
  },
138
  {
 
141
  "shader": "dynamic-quantize-linear.wgsl.jinja",
142
  "derive": { "fromPartials": true },
143
  "bindings": [
144
+ "partial_min_f32",
145
+ "partial_max_f32",
146
  "y_scale",
147
  "y_zero_point",
148
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "fullPartials" }] }
 
153
  "id": "quantize",
154
  "name": "DynamicQuantizeLinear.Quantize",
155
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
156
+ "bindings": ["x_reduce", "y_scale_f32", "y_zero_point_u32", "y", "params"],
157
  "dispatch": {
158
+ "x": "min(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
159
+ "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
160
  "z": 1
161
  }
162
  }
 
182
  "id": "reduce",
183
  "name": "DynamicQuantizeLinear.ReduceMinMax",
184
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
 
185
  "derive": { "gridStride": true },
186
+ "bindings": ["x_reduce", "partial_min", "partial_max", "params"],
187
  "dispatch": { "x": "gridPartials" }
188
  },
189
  {
 
192
  "shader": "dynamic-quantize-linear.wgsl.jinja",
193
  "derive": { "fromPartials": true },
194
  "bindings": [
195
+ "partial_min_f32",
196
+ "partial_max_f32",
197
  "y_scale",
198
  "y_zero_point",
199
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "gridPartials" }] }
 
204
  "id": "quantize",
205
  "name": "DynamicQuantizeLinear.Quantize",
206
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
207
+ "bindings": ["x_reduce", "y_scale_f32", "y_zero_point_u32", "y", "params"],
208
  "dispatch": {
209
+ "x": "min(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
210
+ "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
211
  "z": 1
212
  }
213
  }
 
233
  "id": "reduce",
234
  "name": "DynamicQuantizeLinear.ReduceMinMax",
235
  "shader": "dynamic-quantize-linear-reduce.wgsl.jinja",
 
236
  "derive": { "gridStride": true },
237
+ "bindings": ["x_reduce", "partial_min", "partial_max", "params"],
238
  "dispatch": { "x": "gridPartials" }
239
  },
240
  {
 
243
  "shader": "dynamic-quantize-linear.wgsl.jinja",
244
  "derive": { "fromPartials": true },
245
  "bindings": [
246
+ "partial_min_f32",
247
+ "partial_max_f32",
248
  "y_scale",
249
  "y_zero_point",
250
  { "name": "params", "struct": [{ "name": "numPartials", "type": "u32", "value": "gridPartials" }] }
 
255
  "id": "quantize",
256
  "name": "DynamicQuantizeLinear.Quantize",
257
  "shader": "dynamic-quantize-linear-quantize.wgsl.jinja",
258
+ "bindings": ["x_reduce", "y_scale_f32", "y_zero_point_u32", "y", "params"],
259
  "dispatch": {
260
+ "x": "min(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
261
+ "y": "ceilDiv(ceilDiv((ceilDiv(inputCount, 4 if vec4 else tunables.ELEMENTS_PER_THREAD)), (tunables.WORKGROUP_SIZE)), 65535)",
262
  "z": 1
263
  }
264
  }
build/webgpu/metadata.json CHANGED
@@ -1,23 +1,23 @@
1
  {
2
  "name": "ai.onnx.DynamicQuantizeLinear",
3
- "id": "_ai_onnx_dynamicquantizelinear_webgpu_c781458",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "backend": { "type": "webgpu" },
7
  "digest": {
8
  "algorithm": "sha256",
9
  "files": {
10
- "bench.json": "XDg7UAukA6OblWGm+ypY2zEJvMbW7c2WSl5jcuO5zxU=",
11
- "dynamic-quantize-linear-quantize.wgsl.jinja": "ISRZ+RP/JwuhPFFtM/mSTJ+qGU8ATQz9B8b1bnZuS18=",
12
- "dynamic-quantize-linear-reduce.wgsl.jinja": "MbIpHB8DKZZo8wVxryLnQkF9UeI5+GDuPb67Wq5CwhA=",
13
- "dynamic-quantize-linear.wgsl.jinja": "tGwmn/Rl6x4gkOe+YNfxsBhJ6TrZgeNFvOXCNfOFeq4=",
14
- "manifest.json": "bndT+hJeO/n9kaxatO81kxGcbg9bdnNYv9uEEOY8SHw=",
15
- "test.json": "moR9c5IlMwac/omRuy88wGQgWfM/KKQRuM9+oU68p1M="
16
  }
17
  },
18
- "provenance": { "kernel": { "sha": "91d990483a174128daf7673f3f37a7c890493ae1", "dirty": false } },
19
  "webgpu": {
20
- "manifestSpec": "2.0",
21
  "variants": {
22
  "single_invocation": ["dynamic-quantize-linear.wgsl.jinja"],
23
  "parallel_subgroup_reduce_vec4": ["dynamic-quantize-linear-quantize.wgsl.jinja", "dynamic-quantize-linear-reduce.wgsl.jinja", "dynamic-quantize-linear.wgsl.jinja"],
 
1
  {
2
  "name": "ai.onnx.DynamicQuantizeLinear",
3
+ "id": "_ai_onnx_dynamicquantizelinear_webgpu_8b415f1",
4
  "version": 1,
5
  "license": "Apache-2.0",
6
  "backend": { "type": "webgpu" },
7
  "digest": {
8
  "algorithm": "sha256",
9
  "files": {
10
+ "bench.json": "qmSAe27A81lhwOldgupcB5tAO91fQaX9ie9NSiZ+TBg=",
11
+ "dynamic-quantize-linear-quantize.wgsl.jinja": "v/Qbmu+9HfN3BrtdMuvmO0e90GIWjYmIyaFg+lWkhvQ=",
12
+ "dynamic-quantize-linear-reduce.wgsl.jinja": "R4xvUVH5tHM7awxfoRAFVnl1RCqLshJgwJHrsCwa/s8=",
13
+ "dynamic-quantize-linear.wgsl.jinja": "+uYFmm3hTfefy/KEkqV/6rCUmLHWYoZBL6i+dLvSDkU=",
14
+ "manifest.json": "4sGueX5KlDKMgz+FsV2CrCjV6WxGzUMpOnXjO3laoxw=",
15
+ "test.json": "2kKDGYaa2O5ih/oJoD/57r0AMEm5PUSJBnm57pNjOxg="
16
  }
17
  },
18
+ "provenance": { "kernel": { "sha": "6fdf6301e2bbcc2f03bf1eaf493b7ad55ef33afc", "dirty": false } },
19
  "webgpu": {
20
+ "manifestSpec": "2.1",
21
  "variants": {
22
  "single_invocation": ["dynamic-quantize-linear.wgsl.jinja"],
23
  "parallel_subgroup_reduce_vec4": ["dynamic-quantize-linear-quantize.wgsl.jinja", "dynamic-quantize-linear-reduce.wgsl.jinja", "dynamic-quantize-linear.wgsl.jinja"],
build/webgpu/test.json CHANGED
@@ -317,7 +317,7 @@
317
  {
318
  "name": "grid_stride_reduce_1m_mixed_sign",
319
  "provenance": {
320
- "notes": "A 1,048,576-element input selects a grid-stride reduction capped at 256 workgroups, each producing one extrema partial. The final fold must include all elements and reproduce the exact global minimum, maximum, scale, and zero point."
321
  },
322
  "inputs": {
323
  "x": {
@@ -427,7 +427,7 @@
427
  }
428
  },
429
  {
430
- "name": "intel_d3d_compensated_zero_point_regression",
431
  "provenance": {
432
  "notes": "An explicit expected output at a symmetric half-step boundary requires the correctly rounded zero point 127 rather than 128."
433
  },
 
317
  {
318
  "name": "grid_stride_reduce_1m_mixed_sign",
319
  "provenance": {
320
+ "notes": "A 1,048,576-element input checks that every value contributes to the global minimum and maximum, which set the scale and zero point."
321
  },
322
  "inputs": {
323
  "x": {
 
427
  }
428
  },
429
  {
430
+ "name": "symmetric_half_step_zero_point_boundary_127",
431
  "provenance": {
432
  "notes": "An explicit expected output at a symmetric half-step boundary requires the correctly rounded zero point 127 rather than 128."
433
  },