Download build/webgpu/test.json from webgpu-kernels/ai.onnx.ReduceMean: direct link, hf CLI and curl.
- Browser
- Download file 68.6 kB
-
https://huggingface.co/kernels/webgpu-kernels/ai.onnx.ReduceMean/resolve/v1/build/webgpu/test.json
- Command line
-
hf download hf://webgpu-kernels/ai.onnx.ReduceMean@v1/build/webgpu/test.json
-
curl -L -o test.json https://huggingface.co/kernels/webgpu-kernels/ai.onnx.ReduceMean/resolve/v1/build/webgpu/test.json
68.6 kB
| { | |
| "fixtureArrays": { | |
| "rank3_axis2_last_keepdims_input_x": [1, 2, 3, 4, -1, -2, -3, -4, 0.5, 1.5, 2.5, 3.5, 10, 20, 30, 40, -10, -20, -30, -40, 2, 4, 6, 8] | |
| }, | |
| "cases": [ | |
| { | |
| "name": "contiguous_suffix_axes23_parallel", | |
| "attrs": { "axes": [2, 3], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 16, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.07, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 3, 1, 1], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "all_axes_flat_rank1_boundary_8192", | |
| "provenance": { | |
| "notes": "Exactly 8,192 rank-1 elements exercise the inclusive lower boundary of the parallel full reduction." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [8192], "data": { "kind": "constant", "value": 1.0 } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "all_axes_flat_fullreduce_32x32x32_keepdims", | |
| "provenance": { | |
| "notes": "The all-axes route reduces 32,768 elements through f32 partials and a final combine. Values oscillate around 1, making an incorrect final divisor observable despite the absolute tolerance." | |
| }, | |
| "attrs": { "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [32, 32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.5, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 1, 1], "tolerance": 0.0001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "dispatch_cliff_axis1_rank2", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { "dtype": "float32", "shape": [16776961, 1], "data": { "kind": "linspace", "start": -1.0, "end": 1.0 } } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [16776961], "tolerance": 0.0001 } } | |
| }, | |
| { | |
| "name": "axis0", | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 4], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.23, "cosStep": 0.11 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [4], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "axis0_tiled_64x32", | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [64, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [32], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "f32_subnormal_axis0_tilecols_mean_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "An axis-0 mean of equal finite subnormal values remains subnormal in each column." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [64, 16], "data": { "kind": "constant", "value": 1e-40 } } }, | |
| "outputs": { | |
| "y": { "dtype": "float32", "shape": [16], "tolerance": 2e-45, "data": { "kind": "constant", "value": 1e-40 } } | |
| } | |
| }, | |
| { | |
| "name": "axis1", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 4], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.23, "cosStep": 0.11 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "f32_axis1_parallel_cancellation_order_gpu_gap", | |
| "skipGpu": { | |
| "category": "todo", | |
| "reason": "The current parallel reduction changes the fixture's required sequential evaluation order, so f32 rounding is not bit-exact. An order-preserving reduction route can implement this behavior." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "Mean inherits the same cancellation-order trap as ReduceSum: serial float32 summation yields 0, while the parallel row tree can preserve the small lane terms before division." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [1, 1024], | |
| "data": { "kind": "cycle", "values": [100000000000000000000.0, 1.0, -100000000000000000000.0, 0.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "f32_subnormal_axis1_mean_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "A row mean over equal finite subnormal values remains subnormal; reduction kernels must not flush the input or final quotient to zero." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3], | |
| "data": { "kind": "values", "values": [1e-40, 1e-40, 1e-40, -1e-40, -1e-40, -1e-40] } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { | |
| "dtype": "float32", | |
| "shape": [2], | |
| "tolerance": 2e-45, | |
| "data": { "kind": "values", "values": [1e-40, -1e-40] } | |
| } | |
| } | |
| }, | |
| { | |
| "name": "f32_subnormal_axis0_mean_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "An axis-0 mean of equal finite subnormal column values remains subnormal." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2], | |
| "data": { "kind": "values", "values": [1e-40, -1e-40, 1e-40, -1e-40, 1e-40, -1e-40] } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { | |
| "dtype": "float32", | |
| "shape": [2], | |
| "tolerance": 2e-45, | |
| "data": { "kind": "values", "values": [1e-40, -1e-40] } | |
| } | |
| } | |
| }, | |
| { | |
| "name": "f32_subnormal_last_axis_vec4_mean_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "A vectorized last-axis mean of equal finite subnormal values should remain subnormal." | |
| }, | |
| "attrs": { "axes": [-1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 4], | |
| "data": { "kind": "values", "values": [1e-40, 1e-40, 1e-40, 1e-40, -1e-40, -1e-40, -1e-40, -1e-40] } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { | |
| "dtype": "float32", | |
| "shape": [2], | |
| "tolerance": 2e-45, | |
| "data": { "kind": "values", "values": [1e-40, -1e-40] } | |
| } | |
| } | |
| }, | |
| { | |
| "name": "f32_subnormal_last_axis_odd_mean_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "An odd-width last-axis mean of equal finite subnormal values should remain subnormal." | |
| }, | |
| "attrs": { "axes": [-1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3], | |
| "data": { "kind": "values", "values": [1e-40, 1e-40, 1e-40, -1e-40, -1e-40, -1e-40] } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { | |
| "dtype": "float32", | |
| "shape": [2], | |
| "tolerance": 2e-45, | |
| "data": { "kind": "values", "values": [1e-40, -1e-40] } | |
| } | |
| } | |
| }, | |
| { | |
| "name": "f32_subnormal_rank3_axis1_mean_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "A rank-3 axis-1 mean of equal finite subnormal values remains subnormal through middle-axis indexing." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 2], | |
| "data": { | |
| "kind": "values", | |
| "values": [1e-40, -1e-40, 1e-40, -1e-40, 1e-40, -1e-40, -1e-40, 1e-40, -1e-40, 1e-40, -1e-40, 1e-40] | |
| } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { | |
| "dtype": "float32", | |
| "shape": [2, 2], | |
| "tolerance": 2e-45, | |
| "data": { "kind": "values", "values": [1e-40, -1e-40, -1e-40, 1e-40] } | |
| } | |
| } | |
| }, | |
| { | |
| "name": "f32_subnormal_rank3_all_axes_mean_scalar_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_default_axes_do_not_keep_dims", | |
| "notes": "A rank-3 default-axes mean of equal finite subnormal values should remain subnormal in scalar output form." | |
| }, | |
| "attrs": { "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [2, 3, 2], "data": { "kind": "constant", "value": 1e-40 } } }, | |
| "outputs": { | |
| "y": { "dtype": "float32", "shape": [], "tolerance": 2e-45, "data": { "kind": "values", "values": [1e-40] } } | |
| } | |
| }, | |
| { | |
| "name": "f32_subnormal_rank3_all_axes_keepdims_mean_gpu_gap", | |
| "skipGpu": { | |
| "category": "permanent", | |
| "reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; bit-exact subnormal preservation is unattainable on GPU." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_default_axes_keepdims", | |
| "notes": "A rank-3 default-axes mean with keepdims should remain subnormal in shape [1,1,1]." | |
| }, | |
| "attrs": { "keepdims": 1 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [2, 3, 2], "data": { "kind": "constant", "value": 1e-40 } } }, | |
| "outputs": { | |
| "y": { | |
| "dtype": "float32", | |
| "shape": [1, 1, 1], | |
| "tolerance": 2e-45, | |
| "data": { "kind": "values", "values": [1e-40] } | |
| } | |
| } | |
| }, | |
| { | |
| "name": "axis1_empty_cols_identity_zero", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [2, 0], "data": { "kind": "values", "values": [] } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "axis0_empty_rows_identity_zero", | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [0, 3], "data": { "kind": "values", "values": [] } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "axis1_zero_rows_noop", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [0, 3], "data": { "kind": "values", "values": [] } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [0], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "axis_minus_one", | |
| "attrs": { "axes": [-1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 4], | |
| "data": { "kind": "values", "values": [1.0, 2.0, 3.0, 4.0, -1.0, -2.0, -3.0, -4.0, 0.5, 1.5, 2.5, 3.5] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "rank3_axis2_last_keepdims", | |
| "attrs": { "axes": [2], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 4], | |
| "data": { "kind": "values", "values": { "$ref": "#/fixtureArrays/rank3_axis2_last_keepdims_input_x" } } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 3, 1], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "rank4_axis1_channel_no_keepdims", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 2, 2], | |
| "data": { | |
| "kind": "values", | |
| "values": [1.0, -2.0, 3.0, -4.0, 10.0, 20.0, -30.0, -40.0, 0.25, -0.5, 0.75, -1.0, -5.0, 6.0, -7.0, 8.0, 0.0, 0.0, 1.5, -1.5, 100.0, -200.0, 300.0, -400.0] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 2, 2], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "rank1_axis0_scalar_output", | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [6], | |
| "data": { "kind": "values", "values": [1.0, -2.0, 3.5, 4.5, -1.0, 0.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "rank3_axis2_empty_axis_identity_zero", | |
| "attrs": { "axes": [2], "keepdims": 1 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [2, 3, 0], "data": { "kind": "values", "values": [] } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 3, 1], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "rank4_axis1_empty_axis_identity_zero", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [2, 0, 2, 2], "data": { "kind": "values", "values": [] } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 2, 2], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "ort_axis1_rank3_no_keepdims", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_do_not_keepdims" | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 2], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "ort_axis1_rank3_keepdims", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_keepdims" | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 1, 2], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "ort_axis0_rank1_scalar", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_do_not_keepdims_2" | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [3], "data": { "kind": "values", "values": [1.0, 2.0, 3.0] } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "ort_rank0_scalar", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean0DTensor" | |
| }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [], "data": { "kind": "values", "values": [2.0] } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "ort_axis0_singleton_keepdims_noop", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_keepdims_results_in_noop" | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { "dtype": "float32", "shape": [1, 3], "data": { "kind": "values", "values": [1.0, 2.0, 3.0] } } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 3], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "ort_axis0_singleton_no_keepdims_shape_change", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_keepdims_results_in_shape_change" | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { "dtype": "float32", "shape": [1, 3], "data": { "kind": "values", "values": [1.0, 2.0, 3.0] } } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "ort_default_axes_rank3_no_keepdims_scalar", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_default_axes_do_not_keep_dims", | |
| "notes": "Default axes reduce all input dimensions to a rank-0 scalar when keepdims=0." | |
| }, | |
| "attrs": { "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_do_not_keepdims_example", | |
| "attrs": { "keepdims": 0, "axes": [1] }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 2] } }, | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_do_not_keepdims_example", | |
| "notes": "The ONNX int64 axes input is materialized as this compile-time axes list." | |
| } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_do_not_keepdims_random", | |
| "attrs": { "keepdims": 0, "axes": [1] }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { | |
| "kind": "values", | |
| "values": [0.9762700796127319, 4.3037872314453125, 2.055267572402954, 0.8976636528968811, -1.5269039869308472, 2.917882204055786, -1.248255729675293, 7.835460186004639, 9.273255348205566, -2.331169605255127, 5.834500789642334, 0.577898383140564] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 2] } }, | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_do_not_keepdims_random", | |
| "notes": "The ONNX int64 axes input is materialized as this compile-time axes list." | |
| } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_keepdims_example", | |
| "attrs": { "keepdims": 1, "axes": [1] }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 1, 2] } }, | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_keepdims_example", | |
| "notes": "The ONNX int64 axes input is materialized as this compile-time axes list." | |
| } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_keepdims_random", | |
| "attrs": { "keepdims": 1, "axes": [1] }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { | |
| "kind": "values", | |
| "values": [0.9762700796127319, 4.3037872314453125, 2.055267572402954, 0.8976636528968811, -1.5269039869308472, 2.917882204055786, -1.248255729675293, 7.835460186004639, 9.273255348205566, -2.331169605255127, 5.834500789642334, 0.577898383140564] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 1, 2] } }, | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_keepdims_random", | |
| "notes": "The ONNX int64 axes input is materialized as this compile-time axes list." | |
| } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_negative_axes_keepdims_example", | |
| "attrs": { "keepdims": 1, "axes": [-2] }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 1, 2] } }, | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_negative_axes_keepdims_example", | |
| "notes": "The ONNX int64 axes input is materialized as this compile-time axes list." | |
| } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_negative_axes_keepdims_random", | |
| "attrs": { "keepdims": 1, "axes": [-2] }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { | |
| "kind": "values", | |
| "values": [0.9762700796127319, 4.3037872314453125, 2.055267572402954, 0.8976636528968811, -1.5269039869308472, 2.917882204055786, -1.248255729675293, 7.835460186004639, 9.273255348205566, -2.331169605255127, 5.834500789642334, 0.577898383140564] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 1, 2] } }, | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_negative_axes_keepdims_random", | |
| "notes": "The ONNX int64 axes input is materialized as this compile-time axes list." | |
| } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_default_axes_keepdims_example", | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_default_axes_keepdims_example" | |
| }, | |
| "attrs": { "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 1, 1] } } | |
| }, | |
| { | |
| "name": "ort_default_axes_keepdims_all_rank3", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_default_axes_keepdims" | |
| }, | |
| "attrs": { "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 1, 1], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "onnx_backend_reduce_mean_default_axes_keepdims_random", | |
| "provenance": { | |
| "source": "cmake/external/onnx/onnx/backend/test/data/node/test_reduce_mean_default_axes_keepdims_random" | |
| }, | |
| "attrs": { "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { | |
| "kind": "values", | |
| "values": [0.9762700796127319, 4.3037872314453125, 2.055267572402954, 0.8976636528968811, -1.5269039869308472, 2.917882204055786, -1.248255729675293, 7.835460186004639, 9.273255348205566, -2.331169605255127, 5.834500789642334, 0.577898383140564] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 1, 1] } } | |
| }, | |
| { | |
| "name": "subgroup_vec4_last_axis_2x256", | |
| "attrs": { "axes": [-1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 256], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2], "tolerance": 0.0002, "relTolerance": 0.0001 } } | |
| }, | |
| { | |
| "name": "subgroup_scalar_last_axis_2x65", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 65], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.23, "cosStep": 0.11 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2], "tolerance": 0.0002, "relTolerance": 0.0001 } } | |
| }, | |
| { | |
| "name": "ort_int32_large_values_no_overflow_gpu_gap", | |
| "skipGpu": { | |
| "category": "todo", | |
| "reason": "The current integer reduction route uses an i32 accumulator, so the fixture's 6e9 intermediate sum overflows before division. A portable multiword accumulator can implement this behavior." | |
| }, | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_int32_LargeValues_NoOverflow" | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "int32", | |
| "shape": [3], | |
| "data": { "kind": "values", "values": [2000000000, 2000000000, 2000000000] } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { "dtype": "int32", "shape": [1], "data": { "kind": "values", "values": [2000000000] }, "tolerance": 0 } | |
| } | |
| }, | |
| { | |
| "name": "ort_noop_empty_axes_identity", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_noop_axes_input_initializer_opset_18", | |
| "notes": "The omitted axes input exercises empty-axes behavior." | |
| }, | |
| "attrs": { "keepdims": 0, "noop_with_empty_axes": 1 }, | |
| "inputs": { | |
| "x": { "dtype": "float32", "shape": [1, 2, 2], "data": { "kind": "values", "values": [1.0, 2.0, 3.0, 4.0] } } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 2, 2], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "ort_int32_multi_axis_keepdims", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_int32" | |
| }, | |
| "attrs": { "axes": [0, 2], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "int32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [10, 20, 30, 40, 50, 60, 70, 80, 90, 100, 110, 120] } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { "dtype": "int32", "shape": [1, 2, 1], "data": { "kind": "values", "values": [55, 75] }, "tolerance": 0 } | |
| } | |
| }, | |
| { | |
| "name": "ort_float_multi_axis_keepdims", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean" | |
| }, | |
| "attrs": { "axes": [0, 2], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0, 12.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 2, 1], "tolerance": 0.000001 } } | |
| }, | |
| { | |
| "name": "rank3_lastaxis_cols1024_tree_nosubgroup", | |
| "attrs": { "axes": [2], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 2, 1024], | |
| "data": { | |
| "kind": "cycle", | |
| "values": [1.0, -2.0, 0.5, 3.25, -1.5, 2.0, -0.75, 4.0, -3.5, 1.25, 0.0, -2.25, 5.0, -4.0, 2.75, -1.0, 6.5] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 2], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "axis0_splitk_8192x32", | |
| "provenance": { | |
| "notes": "An offset input keeps each mean over 8,192 rows near one rather than cancelling toward zero. This makes an incorrect axis-length divisor, missing final division, or double-counted partial observable." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [8192, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [32], "tolerance": 0.0001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "axis0_splitk_8192x48_keepdims", | |
| "provenance": { | |
| "notes": "An 8,192-by-48 axis-0 mean with keepdims uses split partials and a final combine. Values oscillate around 1, making an incorrect divisor observable as well as the retained output shape." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [8192, 48], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 48], "tolerance": 0.0001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "f32_rank4_axis2_no_keepdims", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "Rank-4 single-axis reduce over a middle (non-last, non-axis1) dimension." | |
| }, | |
| "attrs": { "axes": [2], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 4, 2], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.37, "cosStep": 0.13, "scale": 0.5 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 3, 2], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "f32_rank4_multi_axis_23_keepdims", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "Rank-4 multi-axis reduce over the trailing spatial axes [2,3]." | |
| }, | |
| "attrs": { "axes": [2, 3], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 4, 2], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.29, "cosStep": 0.17, "scale": 0.5 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 3, 1, 1], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "f32_rank4_default_all_axes_scalar", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean_default_axes_do_not_keep_dims", | |
| "notes": "Reduces every axis of a rank-4 float32 tensor to a rank-0 scalar." | |
| }, | |
| "attrs": { "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 4, 2], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.41, "cosStep": 0.19, "scale": 0.5 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "f32_last_axis_inf_nan_propagation", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "A vectorized last-axis reduction must map an all-positive-infinity row to positive infinity, mixed positive/negative infinities to NaN, and any row containing NaN to NaN." | |
| }, | |
| "attrs": { "axes": [-1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [4, 4], | |
| "data": { | |
| "kind": "values", | |
| "values": [1.0, 2.0, 3.0, 4.0, "Infinity", 1.0, 2.0, 3.0, "Infinity", "-Infinity", 1.0, 1.0, "NaN", 1.0, 2.0, 3.0] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [4], "tolerance": 0.000001, "allowNaN": true } } | |
| }, | |
| { | |
| "name": "rank3_multi_axes_12_keepdims", | |
| "provenance": { | |
| "source": "onnxruntime/test/providers/cpu/reduction/reduction_ops_test.cc", | |
| "test": "ReductionOpTest.ReduceMean", | |
| "notes": "A rank-3 mean over axes [1,2] with keepdims exercises the corresponding multi-axis mask." | |
| }, | |
| "attrs": { "axes": [1, 2], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 4], | |
| "data": { "kind": "values", "values": { "$ref": "#/fixtureArrays/rank3_axis2_last_keepdims_input_x" } } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 1, 1], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "int32_mean_truncation_toward_zero", | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { "dtype": "int32", "shape": [2, 4], "data": { "kind": "values", "values": [-3, 5, -7, 9, -4, 2, -8, 2] } } | |
| }, | |
| "outputs": { | |
| "y": { "dtype": "int32", "shape": [4], "data": { "kind": "values", "values": [-3, 3, -7, 5] }, "tolerance": 0 } | |
| } | |
| }, | |
| { | |
| "name": "reduce_size1_axis_returns_input_value", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 1, 4], | |
| "data": { "kind": "values", "values": [1.0, -2.0, 3.0, -4.0, 5.0, -6.0, 7.0, -8.0] } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { | |
| "dtype": "float32", | |
| "shape": [2, 4], | |
| "data": { "kind": "values", "values": [1.0, -2.0, 3.0, -4.0, 5.0, -6.0, 7.0, -8.0] }, | |
| "tolerance": 0 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "all_axes_flat_numel_boundary_8192", | |
| "provenance": { | |
| "notes": "Exactly 8,192 elements exercise the lower boundary of the split full reduction. Values oscillate around 1, making confusion between the partial count and element count observable." | |
| }, | |
| "attrs": { "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [1, 8192], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.01, "cosStep": 0.02, "scale": 1.0, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [], "tolerance": 0.0001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "axis0_narrow_f32_8192x3_splitk", | |
| "provenance": { | |
| "notes": "An 8,192-by-3 axis-0 mean exercises split-K with a narrow output. Constant ones verify that finalization remains exactly one." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float32", "shape": [8192, 3], "data": { "kind": "constant", "value": 1.0 } } }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "contiguous_suffix_axes12_parallel", | |
| "provenance": { | |
| "notes": "Contiguous axes {1,2} exercise the shared cooperative suffix reduction instead of one serial lane per output." | |
| }, | |
| "attrs": { "axes": [1, 2], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 16, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.07, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 1, 1], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "rank3_axis1_tiled_middle_reduction", | |
| "provenance": { | |
| "notes": "Compact route fixture for the coalesced multi-lane middle-axis reduction used by the 8x1024x768 case." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 64, 64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.037, "cosStep": 0.061, "scale": 0.7 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 64], "tolerance": 0.00002 } } | |
| }, | |
| { | |
| "name": "axis_split_rank3_axis1_2x8192x4", | |
| "provenance": { | |
| "notes": "A rank-3 axis-1 mean over 8,192 elements writes split partials and divides in the combine pass. Values oscillate around 1, making an incorrect partial or final divisor observable under the declared tolerance." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 8192, 4], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 4], "tolerance": 0.0001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "f16_axis_split_tiled_narrow_2x8192x4", | |
| "provenance": { | |
| "notes": "Offsetting the input around 1.0 keeps the mean at O(1), so the tightened tolerance detects errors in the float16 partial representation and the axis-split combine divisor at roughly two float16 ulps." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [2, 8192, 4], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [2, 4], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "axis_split_rank3_axis1_wide_2x8192x32", | |
| "provenance": { | |
| "notes": "The wide axis_split route (inner 32, not the tiled-narrow path) averages zero-mean data to 7e-5 against a 1e-4 tolerance (min detectable uniform scale error 1.43). Offsetting x about 1.0 makes the mean O(1) so the wide route's split count, partial stride, and final divide are all under test." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 8192, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 32], "tolerance": 0.0001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "f16_axis_split_wide_2x8192x32", | |
| "provenance": { | |
| "notes": "f16 wide axis_split: 0.05 absolute tolerance against a 7e-5 mean is a min detectable uniform scale error of 710, so nothing multiplicative is observable on this route. Offsetting x about 1.0 makes the mean O(1) with an f16-resolution tolerance." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [2, 8192, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [2, 32], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_rank3_axis1_serial", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [3, 2, 2], | |
| "data": { "kind": "values", "values": [5.0, 1.0, 20.0, 2.0, 30.0, 1.0, 40.0, 2.0, 55.0, 1.0, 60.0, 2.0] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [3, 2], "tolerance": 0.02 } } | |
| }, | |
| { | |
| "name": "f16_last_axis_serial_fallback", | |
| "provenance": { | |
| "notes": "A 65-column f16 row uses a 128-lane reduction geometry and a scalar tail. Values oscillate around 1 so the mean magnitude remains observable under the f16 tolerance." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [2, 65], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.23, "cosStep": 0.11, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [2], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_all_axes", | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { "x": { "dtype": "float16", "shape": [8192], "data": { "kind": "constant", "value": 1.0 } } }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [], "tolerance": 0.02 } } | |
| }, | |
| { | |
| "name": "f16_axis0_splitk_8192x8", | |
| "provenance": { | |
| "notes": "An 8,192-by-8 f16 axis-0 mean exercises split partials and a final combine. Values oscillate around 1, making an incorrect element-count divisor visible under the f16 tolerance." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [8192, 8], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [8], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_last_axis_vec4_8x1024", | |
| "provenance": { | |
| "notes": "Eight f16 rows of width 1,024 exercise vectorized last-axis reduction. Values oscillate around 1, making confusion between the vector count and column count observable." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [8, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [8], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_last_axis_scalar_8x1023", | |
| "provenance": { | |
| "notes": "Eight f16 rows of width 1,023 exercise scalar last-axis reduction with a ragged lane tail. Values oscillate around 1 so using 1,024 as the divisor or miscounting the tail is observable." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [8, 1023], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [8], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_all_axes_flat_65543", | |
| "provenance": { | |
| "notes": "A 65,543-element f16 full reduction has a ragged final workgroup span. Values oscillate around 1 so double-counting the tail or dividing by a padded length is observable." | |
| }, | |
| "attrs": { "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [65543], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_suffix_vec4_4x8x128", | |
| "provenance": { | |
| "notes": "Each row reduces a contiguous 1,024-element f16 suffix with vectorized loads. Values oscillate around 1, making the suffix element count used as the divisor observable." | |
| }, | |
| "attrs": { "axes": [1, 2], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [4, 8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [4], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_suffix_scalar_4x7x37", | |
| "provenance": { | |
| "notes": "Each row reduces a non-vectorizable f16 suffix of 259 elements. Values oscillate around 1, so a divisor taken from a padded suffix length instead of 7*37 is observable." | |
| }, | |
| "attrs": { "axes": [1, 2], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [4, 7, 37], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [4], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "f16_axis0_tilecols_4096x64", | |
| "provenance": { | |
| "notes": "Each column mean reduces 4,096 f16 rows in a workgroup. Values oscillate around 1, making confusion between the tile width and row count observable." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [4096, 64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [64], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "int32_axis0_tiled_64x32", | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "int32", | |
| "shape": [64, 32], | |
| "data": { "kind": "cycle", "values": [16777217, 3, -5, 16777219, 7, -11, 2] } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "int32", "shape": [32], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "subgroup_rows_last_axis_f32_96x256", | |
| "attrs": { "axes": [-1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [96, 256], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [96], "tolerance": 0.0002, "relTolerance": 0.0001 } } | |
| }, | |
| { | |
| "name": "subgroup_rows_last_axis_f16_80x1024", | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [80, 1024], | |
| "data": { | |
| "kind": "cycle", | |
| "values": [1.0, 1.1808, 1.3371, 1.4478, 1.4979, 1.4806, 1.3983, 1.262, 1.0903, 0.9064, 0.7351, 0.5997, 0.5184, 0.5024, 0.5537, 0.6654, 0.8224, 1.0034, 1.184, 1.3397, 1.4494, 1.4982, 1.4797, 1.3962, 1.2591, 1.0869, 0.903, 0.7322, 0.5976, 0.5175, 0.5027, 0.5552, 0.6679, 0.8256, 1.0068, 1.1871, 1.3421, 1.4508, 1.4985, 1.4787, 1.3941, 1.2562, 1.0836, 0.8997, 0.7293, 0.5956, 0.5166, 0.5031, 0.5568, 0.6705, 0.8288, 1.0102, 1.1903, 1.3446, 1.4523, 1.4988, 1.4777, 1.392, 1.2533, 1.0802, 0.8963, 0.7265, 0.5936, 0.5158, 0.5035, 0.5584, 0.673, 0.832, 1.0136, 1.1934, 1.3471, 1.4537, 1.499, 1.4767, 1.3899, 1.2503, 1.0769, 0.893, 0.7236, 0.5917, 0.5149, 0.5039, 0.56, 0.6756, 0.8352, 1.017, 1.1966, 1.3495, 1.4552, 1.4992, 1.4757, 1.3878, 1.2474, 1.0735, 0.8897, 0.7208, 0.5897, 0.5141, 0.5043, 0.5616, 0.6782, 0.8384, 1.0204, 1.1997, 1.352, 1.4566, 1.4994, 1.4746, 1.3856, 1.2444, 1.0701, 0.8864, 0.718, 0.5878, 0.5133, 0.5048, 0.5633, 0.6808, 0.8416, 1.0238, 1.2028, 1.3544, 1.4579, 1.4995, 1.4735, 1.3834, 1.2414, 1.0667, 0.883, 0.7152, 0.5858, 0.5126, 0.5053, 0.5649, 0.6835, 0.8449, 1.0272, 1.2059, 1.3568, 1.4593, 1.4997, 1.4724, 1.3812, 1.2384, 1.0634, 0.8797, 0.7124, 0.5839, 0.5118, 0.5058, 0.5666, 0.6861, 0.8481, 1.0306, 1.209, 1.3591, 1.4606, 1.4998, 1.4713, 1.379, 1.2354, 1.06, 0.8764, 0.7096, 0.5821, 0.5111, 0.5063, 0.5683, 0.6888, 0.8514, 1.034, 1.2121, 1.3615, 1.4619, 1.4999, 1.4701, 1.3768, 1.2324, 1.0566, 0.8731, 0.7068, 0.5802, 0.5104, 0.5069, 0.5701, 0.6915, 0.8546, 1.0374, 1.2152, 1.3638, 1.4632, 1.4999, 1.469, 1.3745, 1.2294, 1.0532, 0.8698, 0.7041, 0.5784, 0.5097, 0.5074, 0.5718, 0.6941, 0.8579, 1.0408, 1.2183, 1.3662, 1.4645, 1.5, 1.4678, 1.3723, 1.2264, 1.0498, 0.8665, 0.7013, 0.5765, 0.5091, 0.508, 0.5736, 0.6968, 0.8611, 1.0442, 1.2213, 1.3685, 1.4658, 1.5, 1.4666, 1.37, 1.2233, 1.0464, 0.8633, 0.6986, 0.5747, 0.5084, 0.5086, 0.5754, 0.6996, 0.8644, 1.0476, 1.2244, 1.3708, 1.467, 1.5, 1.4653, 1.3677, 1.2203, 1.043, 0.86, 0.6959, 0.573, 0.5078, 0.5093, 0.5772, 0.7023, 0.8677, 1.051, 1.2274, 1.3731, 1.4682, 1.5, 1.4641, 1.3654, 1.2172, 1.0396, 0.8567, 0.6932, 0.5712, 0.5072, 0.5099, 0.579, 0.705, 0.871, 1.0544, 1.2305, 1.3753, 1.4694, 1.4999, 1.4628, 1.363, 1.2141, 1.0362, 0.8535, 0.6905, 0.5694, 0.5067, 0.5106, 0.5809, 0.7078, 0.8743, 1.0578, 1.2335, 1.3776, 1.4705, 1.4998, 1.4615, 1.3607, 1.211, 1.0328, 0.8502, 0.6878, 0.5677, 0.5061, 0.5113, 0.5827, 0.7106, 0.8776, 1.0612, 1.2365, 1.3798, 1.4717, 1.4997, 1.4602, 1.3583, 1.2079, 1.0294, 0.847, 0.6852, 0.566, 0.5056, 0.5121, 0.5846, 0.7134, 0.8809, 1.0646, 1.2395, 1.382, 1.4728, 1.4996, 1.4588, 1.3559, 1.2048, 1.026, 0.8437, 0.6825, 0.5643, 0.5051, 0.5128, 0.5865, 0.7162, 0.8842, 1.0679, 1.2425, 1.3842, 1.4739, 1.4995, 1.4575, 1.3535, 1.2017, 1.0226, 0.8405, 0.6799, 0.5627, 0.5046, 0.5136, 0.5884, 0.719, 0.8875, 1.0713, 1.2454, 1.3864, 1.475, 1.4993, 1.4561, 1.3511, 1.1986, 1.0192, 0.8373, 0.6773, 0.561, 0.5042, 0.5144, 0.5904, 0.7218, 0.8908, 1.0747, 1.2484, 1.3885, 1.476, 1.4991, 1.4547, 1.3487, 1.1955, 1.0158, 0.834, 0.6747, 0.5594, 0.5037, 0.5152, 0.5923, 0.7246, 0.8942, 1.078, 1.2514, 1.3906, 1.4771, 1.4989, 1.4532, 1.3462, 1.1923, 1.0124, 0.8308, 0.6721, 0.5578, 0.5033, 0.5161, 0.5943, 0.7275, 0.8975, 1.0814, 1.2543, 1.3928, 1.4781, 1.4987, 1.4518, 1.3437, 1.1892, 1.009, 0.8276, 0.6696, 0.5562, 0.503, 0.517, 0.5963, 0.7303, 0.9008, 1.0848, 1.2572, 1.3949, 1.4791, 1.4984, 1.4503, 1.3413, 1.186, 1.0056, 0.8244, 0.667, 0.5547, 0.5026, 0.5178, 0.5983, 0.7332, 0.9042, 1.0881, 1.2601, 1.3969, 1.48, 1.4981, 1.4488, 1.3388, 1.1829, 1.0022, 0.8213, 0.6645, 0.5531, 0.5023, 0.5188, 0.6004, 0.7361, 0.9075, 1.0915, 1.263, 1.399, 1.481, 1.4978, 1.4473, 1.3363, 1.1797, 0.9988, 0.8181, 0.662, 0.5516, 0.502, 0.5197, 0.6024, 0.739, 0.9109, 1.0948, 1.2659, 1.4011, 1.4819, 1.4975, 1.4458, 1.3337, 1.1765, 0.9954, 0.8149, 0.6595, 0.5501, 0.5017, 0.5207, 0.6045, 0.7419, 0.9142, 1.0982, 1.2688, 1.4031, 1.4828, 1.4971, 1.4442, 1.3312, 1.1733, 0.992, 0.8117, 0.657, 0.5486, 0.5014, 0.5216, 0.6066, 0.7448, 0.9176, 1.1015, 1.2717, 1.4051, 1.4837, 1.4968, 1.4427, 1.3286, 1.1701, 0.9886, 0.8086, 0.6545, 0.5472, 0.5012, 0.5226, 0.6087, 0.7478, 0.921, 1.1048, 1.2745, 1.4071, 1.4845, 1.4964, 1.4411, 1.326, 1.1669, 0.9852, 0.8054, 0.6521, 0.5458, 0.5009, 0.5237, 0.6109, 0.7507, 0.9243, 1.1082, 1.2774, 1.409, 1.4853, 1.496, 1.4394, 1.3235, 1.1637, 0.9818, 0.8023, 0.6496, 0.5443, 0.5007, 0.5247, 0.613, 0.7537, 0.9277, 1.1115, 1.2802, 1.411, 1.4862, 1.4955, 1.4378, 1.3208, 1.1605, 0.9784, 0.7992, 0.6472, 0.5429, 0.5006, 0.5258, 0.6152, 0.7567, 0.9311, 1.1148, 1.283, 1.4129, 1.4869, 1.495, 1.4362, 1.3182, 1.1572, 0.975, 0.7961, 0.6448, 0.5416, 0.5004, 0.5269, 0.6174, 0.7596, 0.9344, 1.1181, 1.2858, 1.4148, 1.4877, 1.4946, 1.4345, 1.3156, 1.154, 0.9716, 0.793, 0.6424, 0.5402, 0.5003, 0.528, 0.6196, 0.7626, 0.9378, 1.1214, 1.2886, 1.4167, 1.4884, 1.494, 1.4328, 1.3129, 1.1507, 0.9682, 0.7899, 0.64, 0.5389, 0.5002, 0.5291, 0.6218, 0.7656, 0.9412, 1.1247, 1.2914, 1.4186, 1.4892, 1.4935, 1.4311, 1.3103, 1.1475, 0.9648, 0.7868, 0.6377, 0.5376, 0.5001, 0.5303, 0.624, 0.7686, 0.9446, 1.128, 1.2942, 1.4205, 1.4898, 1.4929, 1.4293, 1.3076, 1.1442, 0.9614, 0.7837, 0.6353, 0.5363, 0.5, 0.5315, 0.6263, 0.7717, 0.948, 1.1313, 1.2969, 1.4223, 1.4905, 1.4924, 1.4276, 1.3049, 1.141, 0.958, 0.7806, 0.633, 0.535, 0.5, 0.5327, 0.6285, 0.7747, 0.9514, 1.1346, 1.2996, 1.4241, 1.4912, 1.4918, 1.4258, 1.3022, 1.1377, 0.9546, 0.7776, 0.6307, 0.5338, 0.5, 0.5339, 0.6308, 0.7778, 0.9548, 1.1379, 1.3024, 1.4259, 1.4918, 1.4911, 1.424, 1.2995, 1.1344, 0.9512, 0.7745, 0.6284, 0.5326, 0.5, 0.5351, 0.6331, 0.7808, 0.9582, 1.1412, 1.3051, 1.4277, 1.4924, 1.4905, 1.4222, 1.2967, 1.1311, 0.9478, 0.7715, 0.6261, 0.5314, 0.5, 0.5364, 0.6355, 0.7839, 0.9616, 1.1444, 1.3078, 1.4294, 1.493, 1.4898, 1.4203, 1.294, 1.1278, 0.9444, 0.7685, 0.6239, 0.5302, 0.5001, 0.5377, 0.6378, 0.787, 0.965, 1.1477, 1.3104, 1.4312, 1.4935, 1.4891, 1.4185, 1.2912, 1.1245, 0.941, 0.7655, 0.6216, 0.529, 0.5002, 0.539, 0.6402, 0.79, 0.9684, 1.1509, 1.3131, 1.4329, 1.4941, 1.4884, 1.4166, 1.2885, 1.1212, 0.9376, 0.7625, 0.6194, 0.5279, 0.5003, 0.5403, 0.6425, 0.7931, 0.9718, 1.1542, 1.3157, 1.4346, 1.4946, 1.4877, 1.4147, 1.2857, 1.1179, 0.9342, 0.7595, 0.6172, 0.5268, 0.5004, 0.5417, 0.6449, 0.7963, 0.9752, 1.1574, 1.3184, 1.4362, 1.4951, 1.4869, 1.4128, 1.2829, 1.1146, 0.9309, 0.7565, 0.615, 0.5257, 0.5006, 0.543, 0.6473, 0.7994, 0.9786, 1.1607, 1.321, 1.4379, 1.4955, 1.4861, 1.4109, 1.28, 1.1113, 0.9275, 0.7535, 0.6129, 0.5246, 0.5007, 0.5444, 0.6498, 0.8025, 0.982, 1.1639, 1.3236, 1.4395, 1.496, 1.4853, 1.4089, 1.2772, 1.108, 0.9241, 0.7506, 0.6107, 0.5236, 0.5009, 0.5458, 0.6522, 0.8056, 0.9854, 1.1671, 1.3262, 1.4412, 1.4964, 1.4845, 1.407, 1.2744, 1.1046, 0.9208, 0.7476, 0.6086, 0.5226, 0.5012, 0.5473, 0.6547, 0.8088, 0.9888, 1.1703, 1.3288, 1.4427, 1.4968, 1.4836, 1.405, 1.2715, 1.1013, 0.9174, 0.7447, 0.6065, 0.5216, 0.5014, 0.5487, 0.6571, 0.8119, 0.9922, 1.1735, 1.3313, 1.4443, 1.4972, 1.4827, 1.403, 1.2686, 1.098, 0.914, 0.7417, 0.6044, 0.5206, 0.5017, 0.5502, 0.6596, 0.8151, 0.9956, 1.1767, 1.3339, 1.4459, 1.4975, 1.4818, 1.4009, 1.2658, 1.0946, 0.9107, 0.7388, 0.6023, 0.5196, 0.502, 0.5517, 0.6621, 0.8183, 0.999, 1.1799, 1.3364, 1.4474, 1.4978, 1.4809, 1.3989, 1.2629, 1.0913, 0.9073, 0.7359, 0.6003, 0.5187, 0.5023, 0.5532, 0.6646, 0.8214, 1.0024, 1.183, 1.3389, 1.4489, 1.4982, 1.48, 1.3968, 1.26, 1.0879, 0.904, 0.733, 0.5982, 0.5178, 0.5026, 0.5548, 0.6672, 0.8246, 1.0058, 1.1862, 1.3414, 1.4504, 1.4984, 1.479, 1.3947, 1.2571, 1.0846, 0.9007, 0.7302, 0.5962, 0.5169, 0.503, 0.5563, 0.6697, 0.8278, 1.0092, 1.1894, 1.3439, 1.4519, 1.4987, 1.478, 1.3926, 1.2541, 1.0812, 0.8973, 0.7273, 0.5942, 0.516, 0.5034, 0.5579, 0.6723, 0.831, 1.0126, 1.1925, 1.3464, 1.4533, 1.4989, 1.477, 1.3905, 1.2512, 1.0779, 0.894, 0.7245, 0.5922, 0.5152, 0.5038, 0.5595, 0.6749, 0.8342, 1.016, 1.1957, 1.3488, 1.4547, 1.4991, 1.476, 1.3884, 1.2482, 1.0745, 0.8907, 0.7216, 0.5903, 0.5144, 0.5042, 0.5611, 0.6775, 0.8374, 1.0194, 1.1988, 1.3512, 1.4562, 1.4993, 1.4749, 1.3862, 1.2453] | |
| } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [80], "tolerance": 0.002, "relTolerance": 0.002 } } | |
| }, | |
| { | |
| "name": "int32_lastaxis_subgroup_rows_64x1024", | |
| "provenance": { | |
| "notes": "Sixty-four rows of 1024 int32 values check integer mean and truncation; a five-value cycle shifts phase so neighboring row results differ (11 or 12)." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { "dtype": "int32", "shape": [64, 1024], "data": { "kind": "cycle", "values": [30, 40, 0, -20, 10] } } | |
| }, | |
| "outputs": { "y": { "dtype": "int32", "shape": [64], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "rank4_axes023_noncontiguous_multi_axis_2x4x3x3", | |
| "provenance": { | |
| "notes": "A rank-4 mean over axes {0,2,3} keeps the middle channel axis. The reduced elements are not a contiguous suffix, so each output uses a serial fold." | |
| }, | |
| "attrs": { "axes": [0, 2, 3], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 4, 3, 3], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.07, "scale": 0.2, "offset": 1.0 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [4], "tolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "multi_axis_rank4_coop_channel_reduce_axes023", | |
| "provenance": { | |
| "notes": "A rank-4 mean over axes {0,2,3} leaves one output per channel and many reduced elements per output, exercising cooperative accumulation." | |
| }, | |
| "attrs": { "axes": [0, 2, 3], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 8, 16, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.029, "scale": 1.5 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [8], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "multi_axis_rank4_coop_channel_reduce_axes023_keepdims", | |
| "provenance": { | |
| "notes": "A rank-4 mean over axes {0,2,3} with keepdims leaves one output per channel and many reduced elements per output, exercising cooperative accumulation." | |
| }, | |
| "attrs": { "axes": [0, 2, 3], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 8, 16, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.029, "scale": 1.5 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [1, 8, 1, 1], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "multi_axis_rank3_coop_axes02", | |
| "provenance": { | |
| "notes": "A rank-3 mean over axes {0,2} leaves one output per channel and many reduced elements per output, exercising cooperative accumulation." | |
| }, | |
| "attrs": { "axes": [0, 2], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [8, 6, 64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.029, "scale": 1.5 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [6], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "multi_axis_rank4_coop_channel_reduce_axes023_i32", | |
| "provenance": { | |
| "notes": "Integer accumulation stays in the output type. A linear ramp gives each retained channel a distinct span and therefore a distinct truncated mean." | |
| }, | |
| "attrs": { "axes": [0, 2, 3], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { "dtype": "int32", "shape": [2, 8, 16, 16], "data": { "kind": "linspace", "start": -2000, "end": 2000 } } | |
| }, | |
| "outputs": { "y": { "dtype": "int32", "shape": [8], "tolerance": 0 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float32_rank3_odd_axis_odd_inner", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [-2], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 127, 259], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 1, 259], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float32_rank3_even_inner", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [7, 129, 260], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [7, 260], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float32_rank5_middle", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [3], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 3, 5, 17, 7], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { | |
| "y": { "dtype": "float32", "shape": [2, 3, 5, 1, 7], "tolerance": 0.00001, "relTolerance": 0.00001 } | |
| } | |
| }, | |
| { | |
| "name": "strided_geometry_float32_rank3_axis0", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [17, 5, 13], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [5, 13], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float32_rank4_middle", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [2], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [3, 5, 17, 4], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [3, 5, 4], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float32_rank4_channel", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float32", | |
| "shape": [2, 63, 7, 8], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float32", "shape": [2, 1, 7, 8], "tolerance": 0.00001, "relTolerance": 0.00001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float16_rank3_odd_axis_odd_inner", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [-2], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [3, 127, 259], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [3, 1, 259], "tolerance": 0.000002, "relTolerance": 0.001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float16_rank3_even_inner", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [7, 129, 260], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [7, 260], "tolerance": 0.000002, "relTolerance": 0.001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float16_rank5_middle", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [3], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [2, 3, 5, 17, 7], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [2, 3, 5, 1, 7], "tolerance": 0.000002, "relTolerance": 0.001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float16_rank3_axis0", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [0], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [17, 5, 13], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [5, 13], "tolerance": 0.000002, "relTolerance": 0.001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float16_rank4_middle", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [2], "keepdims": 0 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [3, 5, 17, 4], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [3, 5, 4], "tolerance": 0.000002, "relTolerance": 0.001 } } | |
| }, | |
| { | |
| "name": "strided_geometry_float16_rank4_channel", | |
| "provenance": { | |
| "notes": "Single-axis flattened addressing across outer and inner dimensions, with signed inputs and distinct outputs. Float16 storage accumulates in float32 and rounds only on the final store." | |
| }, | |
| "attrs": { "axes": [1], "keepdims": 1 }, | |
| "inputs": { | |
| "x": { | |
| "dtype": "float16", | |
| "shape": [2, 63, 7, 8], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.029, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "y": { "dtype": "float16", "shape": [2, 1, 7, 8], "tolerance": 0.000002, "relTolerance": 0.001 } } | |
| } | |
| ] | |
| } | |