Download build/webgpu/manifest.json from webgpu-kernels/com.microsoft.GroupQueryAttention: direct link, hf CLI and curl.
- Browser
- Download file 214 kB
-
https://huggingface.co/kernels/webgpu-kernels/com.microsoft.GroupQueryAttention/resolve/v1/build/webgpu/manifest.json
- Command line
-
hf download hf://webgpu-kernels/com.microsoft.GroupQueryAttention@v1/build/webgpu/manifest.json
-
curl -L -o manifest.json https://huggingface.co/kernels/webgpu-kernels/com.microsoft.GroupQueryAttention/resolve/v1/build/webgpu/manifest.json
214 kB
| { | |
| "domain": "com.microsoft", | |
| "name": "GroupQueryAttention", | |
| "sinceVersion": 1, | |
| "inputs": { | |
| "queryT": { "onnx": "query", "dtype": "T", "rank": 3 }, | |
| "keyT": { "onnx": "key", "dtype": "T", "rank": 3 }, | |
| "valueT": { "onnx": "value", "dtype": "T", "rank": 3 }, | |
| "pastKeyT": { "onnx": "past_key", "dtype": "T_CACHE", "rank": 4, "optional": true }, | |
| "pastValueT": { "onnx": "past_value", "dtype": "T_CACHE", "rank": 4, "optional": true }, | |
| "seqlensKT": { "onnx": "seqlens_k", "dtype": "M", "rank": 1 }, | |
| "totalSequenceLengthT": { "onnx": "total_sequence_length", "dtype": "M", "rank": 1 }, | |
| "cosCacheT": { "onnx": "cos_cache", "dtype": "T", "rank": 2, "optional": true }, | |
| "sinCacheT": { "onnx": "sin_cache", "dtype": "T", "rank": 2, "optional": true }, | |
| "attentionBiasT": { "onnx": "attention_bias", "dtype": "T", "rank": 4, "optional": true }, | |
| "headSinkT": { "onnx": "head_sink", "dtype": "T", "rank": 1, "optional": true }, | |
| "kScaleT": { "onnx": "k_scale", "dtype": "T_KV_SCALE", "rank": 1, "optional": true }, | |
| "vScaleT": { "onnx": "v_scale", "dtype": "T_KV_SCALE", "rank": 1, "optional": true }, | |
| "qNormWeightT": { "onnx": "q_norm_weight", "dtype": "T", "rank": 1, "optional": true }, | |
| "kNormWeightT": { "onnx": "k_norm_weight", "dtype": "T", "rank": 1, "optional": true } | |
| }, | |
| "outputs": { | |
| "outputT": { "onnx": "output", "dtype": "T", "rank": 3, "shape": "shapes.queryT" }, | |
| "presentKeyT": { "onnx": "present_key", "dtype": "T_CACHE", "rank": 4 }, | |
| "presentValueT": { "onnx": "present_value", "dtype": "T_CACHE", "rank": 4 } | |
| }, | |
| "attributes": { | |
| "causal": { "default": 1 }, | |
| "do_rotary": { "default": 0 }, | |
| "k_quant_type": { "default": "NONE" }, | |
| "local_window_size": { "default": -1 }, | |
| "qk_norm_epsilon": { "default": 0.000001 }, | |
| "sliding_window_cache": { "default": 0 }, | |
| "smooth_softmax": { "default": -1 }, | |
| "softcap": { "default": 0 }, | |
| "v_quant_type": { "default": "NONE" }, | |
| "kv_cache_bit_width": {}, | |
| "kv_num_heads": {}, | |
| "num_heads": {}, | |
| "scale": {} | |
| }, | |
| "attributeConstraints": { | |
| "causal": { "values": [0, 1] }, | |
| "do_rotary": { "values": [0, 1] }, | |
| "k_quant_type": { "values": ["NONE", "PER_TENSOR", "PER_CHANNEL"] }, | |
| "kv_cache_bit_width": { "values": [4, 8] }, | |
| "kv_num_heads": { "required": true }, | |
| "num_heads": { "required": true }, | |
| "sliding_window_cache": { "values": [0, 1] }, | |
| "v_quant_type": { "values": ["NONE", "PER_TENSOR", "PER_CHANNEL"] } | |
| }, | |
| "typeConstraints": { | |
| "T": ["float32", "float16"], | |
| "T_CACHE": ["float32", "float16", "uint8", "int8"], | |
| "T_KV_SCALE": ["float32"], | |
| "M": ["int32"] | |
| }, | |
| "tunables": { | |
| "WORKGROUP_SIZE": { "default": 256 }, | |
| "COPY_WORKGROUP_SIZE": { "default": 64 }, | |
| "SCALAR_WORKGROUP_SIZE": { "default": 64 }, | |
| "COOPERATIVE_QUERY_THRESHOLD": { "default": 6144 }, | |
| "MAX_SPLITS": { "default": 16 }, | |
| "QKV_SPLIT_TILE_K": { "default": 128 }, | |
| "CACHED_SPLIT_TILE_K": { "default": 256 }, | |
| "FLASH_MIN_HEAD_DIM": { "default": 32 }, | |
| "FLASH_MAX_HEAD_DIM": { "default": 256 }, | |
| "PREFILL_QUERY_TILE": { "default": 32 }, | |
| "PREFILL_LANES_PER_QUERY": { "default": 8 }, | |
| "PREFILL_QUERY_TILE_F16": { "default": 16 }, | |
| "PREFILL_LANES_PER_QUERY_F16": { "default": 8 }, | |
| "QKV_PREFILL_MIN_QUERY_TOKENS": { "default": 31 }, | |
| "QKV_DECODE_MIN_KV_TOKENS": { "default": 512 }, | |
| "CACHED_FLASH_MIN_HEAD_DIM": { "default": 16 }, | |
| "CACHED_DECODE_MIN_KV_TOKENS": { "default": 1024 }, | |
| "CLUSTER_TILE_K_F32": { "default": 8 }, | |
| "CLUSTER_TILE_K_F16": { "default": 8 }, | |
| "NO_SG_TILE_K_MAX": { "default": 16 }, | |
| "COOPERATIVE_WORKGROUP_SIZE": { "default": 32 }, | |
| "CLUSTER_MAX_SLICE": { "default": 8 }, | |
| "MATERIALIZED_SGMAT_QUERY_TILE": { "default": 64 }, | |
| "MATERIALIZED_SGMAT_KEY_TILE": { "default": 64 }, | |
| "MATERIALIZED_SGMAT_INNER_TILE": { "default": 32 }, | |
| "MATERIALIZED_SOFTMAX_WORKGROUP_SIZE": { "default": 256 }, | |
| "MATERIALIZED_SGMAT_MIN_SEQ": { "default": 256 }, | |
| "MATERIALIZED_SGMAT_PAST_MIN_SEQ": { "default": 256 }, | |
| "MATERIALIZED_SGMAT_WINDOW_MIN_QSEQ": { "default": 128 }, | |
| "MATERIALIZED_SGMAT_APPEND_MIN_QSEQ": { "default": 128 } | |
| }, | |
| "derive": { | |
| "deviceWorkgroupCap": "min(device.limits.maxComputeInvocationsPerWorkgroup, device.limits.maxComputeWorkgroupSizeX)", | |
| "wave32Adapter": "has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize == 32 and device.adapterInfo.subgroupMaxSize == 32", | |
| "subgroupsWave32": "device.features.has(\"subgroups\") and wave32Adapter", | |
| "narrowSubgroupRange": "has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize < device.adapterInfo.subgroupMaxSize and device.adapterInfo.subgroupMaxSize <= 16", | |
| "canPinSubgroupSize32": "device.features.has(\"subgroups\") and device.features.has(\"subgroup-size-control\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize <= 32 and device.adapterInfo.subgroupMaxSize >= 32", | |
| "pinSubgroupSize32": "canPinSubgroupSize32 and not wave32Adapter", | |
| "wave32Effective": "wave32Adapter or pinSubgroupSize32", | |
| "wave32SubgroupsUsable": "subgroupsWave32 or pinSubgroupSize32", | |
| "windowCacheRequested": "attrs.sliding_window_cache == 1", | |
| "headDimPlan": "dim(shapes.queryT, 2) / attrs.num_heads if (ranks.queryT == 3 and attrs.num_heads > 0) else 0", | |
| "copyWorkgroupSize": "min(tunables.COPY_WORKGROUP_SIZE, min(device.limits.maxComputeInvocationsPerWorkgroup, device.limits.maxComputeWorkgroupSizeX))", | |
| "scalarWorkgroupSize": "min(tunables.SCALAR_WORKGROUP_SIZE, min(device.limits.maxComputeInvocationsPerWorkgroup, device.limits.maxComputeWorkgroupSizeX))", | |
| "copyWorkgroupOk": "copyWorkgroupSize >= 1", | |
| "scalarWorkgroupOk": "scalarWorkgroupSize >= 1", | |
| "metadataInputsOk": "present.seqlensKT and present.totalSequenceLengthT", | |
| "unquantizedCacheAttrs": "attrs.k_quant_type == \"NONE\" and attrs.v_quant_type == \"NONE\" and not attrs.kv_cache_bit_width and not present.kScaleT and not present.vScaleT", | |
| "qkNormInputsOk": "present.qNormWeightT == present.kNormWeightT", | |
| "rotaryRequested": "attrs.do_rotary == 1", | |
| "noAuxAttentionInputs": "not present.cosCacheT and not rotaryRequested and not present.qNormWeightT and not present.attentionBiasT and not present.headSinkT", | |
| "standardSoftmax": "attrs.softcap == 0 and attrs.smooth_softmax != 1", | |
| "plainAttentionOptions": "noAuxAttentionInputs and standardSoftmax", | |
| "causalNoWindow": "attrs.causal == 1 and attrs.local_window_size <= 0", | |
| "qkvMaskModeOk": "(attrs.causal == 0 and attrs.local_window_size == -1) or (attrs.causal == 1 and attrs.local_window_size > 0) or causalNoWindow", | |
| "qkvContractOk": "plainAttentionOptions and not windowCacheRequested and copyWorkgroupOk and metadataInputsOk and unquantizedCacheAttrs and qkNormInputsOk and qkvMaskModeOk and (not present.pastKeyT) and (not present.pastValueT) and ranks.queryT == 3 and ranks.keyT == 3 and ranks.valueT == 3 and ranks.outputT == 3 and ranks.presentKeyT == 4 and ranks.presentValueT == 4 and (tensorDtypes.queryT == \"float32\" or tensorDtypes.queryT == \"float16\") and f16Ok(tensorDtypes.queryT) and tensorDtypes.keyT == tensorDtypes.queryT and tensorDtypes.valueT == tensorDtypes.queryT and tensorDtypes.outputT == tensorDtypes.queryT and (tensorDtypes.presentKeyT == \"float32\" or tensorDtypes.presentKeyT == \"float16\") and f16Ok(tensorDtypes.presentKeyT) and tensorDtypes.presentValueT == tensorDtypes.presentKeyT and attrs.num_heads > 0 and attrs.kv_num_heads > 0 and attrs.num_heads % attrs.kv_num_heads == 0 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.keyT, 2) == attrs.kv_num_heads * headDimPlan and dim(shapes.valueT, 2) == dim(shapes.keyT, 2) and dim(shapes.queryT, 0) == dim(shapes.keyT, 0) and dim(shapes.queryT, 0) == dim(shapes.valueT, 0) and dim(shapes.keyT, 1) == dim(shapes.valueT, 1) and dim(shapes.outputT, 0) == dim(shapes.queryT, 0) and dim(shapes.outputT, 1) == dim(shapes.queryT, 1) and dim(shapes.outputT, 2) == dim(shapes.queryT, 2) and dim(shapes.presentKeyT, 0) == dim(shapes.queryT, 0) and dim(shapes.presentKeyT, 1) == attrs.kv_num_heads and dim(shapes.presentKeyT, 2) == dim(shapes.keyT, 1) and dim(shapes.presentKeyT, 3) == headDimPlan and dim(shapes.presentValueT, 0) == dim(shapes.presentKeyT, 0) and dim(shapes.presentValueT, 1) == dim(shapes.presentKeyT, 1) and dim(shapes.presentValueT, 2) == dim(shapes.presentKeyT, 2) and dim(shapes.presentValueT, 3) == dim(shapes.presentKeyT, 3)", | |
| "flashWorkgroupOk": "tunables.WORKGROUP_SIZE <= deviceWorkgroupCap", | |
| "flashShapeOk": "qkvContractOk and flashWorkgroupOk and headDimPlan % 4 == 0 and headDimPlan >= tunables.FLASH_MIN_HEAD_DIM and headDimPlan <= tunables.FLASH_MAX_HEAD_DIM and (dim(shapes.queryT, 1) * attrs.num_heads >= 248 or (dim(shapes.queryT, 1) == 1 and dim(shapes.keyT, 1) >= tunables.QKV_DECODE_MIN_KV_TOKENS) or (dim(shapes.queryT, 1) > 1 and dim(shapes.keyT, 1) >= 2048)) and dim(shapes.queryT, 0) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and dim(shapes.queryT, 1) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", | |
| "decodeSplitKShapeOk": "qkvContractOk and flashWorkgroupOk and dim(shapes.queryT, 1) == 1 and dim(shapes.keyT, 1) >= tunables.QKV_DECODE_MIN_KV_TOKENS and headDimPlan % 4 == 0 and headDimPlan >= tunables.FLASH_MIN_HEAD_DIM and headDimPlan <= tunables.FLASH_MAX_HEAD_DIM and dim(shapes.queryT, 0) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", | |
| "fallbackShapeOk": "qkvContractOk and headDimPlan <= deviceWorkgroupCap and headDimPlan <= tunables.FLASH_MAX_HEAD_DIM", | |
| "gqaHeadDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "qkvScalar": "\"f16\" if tensorDtypes.queryT == \"float16\" else \"f32\"", | |
| "qkvInputVec4": "\"vec4<f16>\" if tensorDtypes.queryT == \"float16\" else \"vec4<f32>\"", | |
| "qkvCacheScalar": "\"f16\" if tensorDtypes.presentKeyT == \"float16\" else \"f32\"", | |
| "qkvCacheVec4": "\"vec4<f16>\" if tensorDtypes.presentKeyT == \"float16\" else \"vec4<f32>\"", | |
| "qkvUsesF16": "tensorDtypes.queryT == \"float16\" or tensorDtypes.presentKeyT == \"float16\"", | |
| "gqaScalar": "\"f16\" if tensorDtypes.queryT == \"float16\" else \"f32\"", | |
| "gqaInputVec4": "\"vec4<f16>\" if tensorDtypes.queryT == \"float16\" else \"vec4<f32>\"", | |
| "qkvNumSplits": "min(tunables.MAX_SPLITS, ceilDiv(dim(shapes.keyT, 1), tunables.QKV_SPLIT_TILE_K))", | |
| "cachedNumSplits": "min(tunables.MAX_SPLITS, ceilDiv(dim(shapes.presentKeyT, 2), tunables.CACHED_SPLIT_TILE_K))", | |
| "cachedBaseOk": "attrs.causal == 1 and metadataInputsOk and qkNormInputsOk", | |
| "sharedKvCacheOk": "cachedBaseOk and unquantizedCacheAttrs and not windowCacheRequested and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) == 0 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim", | |
| "gqaFlashHd": "gqaHeadDim >= tunables.CACHED_FLASH_MIN_HEAD_DIM and gqaHeadDim <= tunables.FLASH_MAX_HEAD_DIM", | |
| "f16OrF32": "(tensorDtypes.queryT == \"float32\" or tensorDtypes.queryT == \"float16\") and f16Ok(tensorDtypes.queryT) and tensorDtypes.keyT == tensorDtypes.queryT and tensorDtypes.valueT == tensorDtypes.queryT and tensorDtypes.outputT == tensorDtypes.queryT and tensorDtypes.presentKeyT == tensorDtypes.queryT and tensorDtypes.presentValueT == tensorDtypes.queryT and ((not present.pastKeyT) or tensorDtypes.pastKeyT == tensorDtypes.queryT) and ((not present.pastValueT) or tensorDtypes.pastValueT == tensorDtypes.queryT)", | |
| "subgroupCluster8": "device.features.has(\"subgroups\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize >= 8 and device.adapterInfo.subgroupMinSize % 8 == 0 and device.adapterInfo.subgroupMaxSize % 8 == 0", | |
| "qkvPrefillQueryTile": "tunables.PREFILL_QUERY_TILE_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_QUERY_TILE", | |
| "qkvPrefillLanesPerQuery": "max(tunables.PREFILL_LANES_PER_QUERY_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_LANES_PER_QUERY, ceilDiv(headDimPlan / 4, tunables.CLUSTER_MAX_SLICE))", | |
| "cachedPrefillQueryTile": "tunables.PREFILL_QUERY_TILE_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_QUERY_TILE", | |
| "biasPrefillQueryTile": "min(cachedPrefillQueryTile, 16) if tensorDtypes.queryT == \"float32\" and not device.features.has(\"subgroups\") and has(device.adapterInfo, \"vendor\") and device.adapterInfo.vendor == \"intel\" else cachedPrefillQueryTile", | |
| "cachedPrefillLanesPerQuery": "max(tunables.PREFILL_LANES_PER_QUERY_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_LANES_PER_QUERY, ceilDiv(gqaHeadDim / 4, tunables.CLUSTER_MAX_SLICE))", | |
| "subgroupQkvPrefillCluster": "not narrowSubgroupRange and device.features.has(\"subgroups\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize >= qkvPrefillLanesPerQuery and device.adapterInfo.subgroupMinSize % qkvPrefillLanesPerQuery == 0 and device.adapterInfo.subgroupMaxSize % qkvPrefillLanesPerQuery == 0", | |
| "subgroupCachedPrefillCluster": "not narrowSubgroupRange and device.features.has(\"subgroups\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize >= cachedPrefillLanesPerQuery and device.adapterInfo.subgroupMinSize % cachedPrefillLanesPerQuery == 0 and device.adapterInfo.subgroupMaxSize % cachedPrefillLanesPerQuery == 0", | |
| "subgroupSplitK": "subgroupCluster8 and device.adapterInfo.subgroupMaxSize <= tunables.WORKGROUP_SIZE and tunables.WORKGROUP_SIZE % device.adapterInfo.subgroupMinSize == 0 and tunables.WORKGROUP_SIZE % device.adapterInfo.subgroupMaxSize == 0", | |
| "gqaQueryCount": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)", | |
| "gqaCooperativeLaneCapacity": "floor((floor(device.limits.maxComputeWorkgroupStorageSize / 16) * 4 - gqaHeadDim) / (gqaHeadDim + 2))", | |
| "gqaCooperativeWorkgroupSize": "min(tunables.COOPERATIVE_WORKGROUP_SIZE, deviceWorkgroupCap, max(1, gqaCooperativeLaneCapacity))", | |
| "cooperativeWorkgroupOk": "gqaCooperativeWorkgroupSize >= 1 and gqaCooperativeWorkgroupSize <= deviceWorkgroupCap", | |
| "gqaCoopWorkgroupBytes": "ceilDiv(((gqaCooperativeWorkgroupSize + 1) * gqaHeadDim + gqaCooperativeWorkgroupSize * 2) * 4, 16) * 16", | |
| "gqaCoop": "cooperativeWorkgroupOk and gqaQueryCount < tunables.COOPERATIVE_QUERY_THRESHOLD and gqaCoopWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "gqaScalarDispatchOk": "gqaCoop or scalarWorkgroupOk", | |
| "gqaDispatchUnits": "gqaQueryCount if gqaCoop else ceilDiv(gqaQueryCount, scalarWorkgroupSize)", | |
| "gqaDispatchFits": "dim(shapes.queryT, 0) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", | |
| "sharedKvFloatOk": "sharedKvCacheOk and f16OrF32", | |
| "newKvPastOk": "cachedBaseOk and unquantizedCacheAttrs and not windowCacheRequested and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) > 0 and f16OrF32 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim and dim(shapes.presentKeyT, 2) == dim(shapes.pastKeyT, 2) + dim(shapes.keyT, 1)", | |
| "windowCapacity": "dim(shapes.pastKeyT, 2) if present.pastKeyT else 0", | |
| "windowFitsCapacity": "attrs.local_window_size <= windowCapacity if attrs.local_window_size > 0 else true", | |
| "windowShiftCacheOk": "cachedBaseOk and unquantizedCacheAttrs and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) > 0 and f16OrF32 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim and windowCacheRequested and dim(shapes.presentKeyT, 2) == windowCapacity and dim(shapes.keyT, 1) <= windowCapacity and windowFitsCapacity", | |
| "windowShiftOk": "windowShiftCacheOk and not present.cosCacheT and not rotaryRequested", | |
| "windowShiftRotaryOk": "windowShiftCacheOk and present.cosCacheT and present.sinCacheT and rotaryRequested and tensorDtypes.queryT == \"float32\" and gqaHeadDim % 2 == 0 and dim(shapes.keyT, 1) == dim(shapes.queryT, 1)", | |
| "shareAppendShapeOk": "unquantizedCacheAttrs and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) > 0 and f16OrF32 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim and not windowCacheRequested and dim(shapes.presentKeyT, 2) == dim(shapes.pastKeyT, 2) and dim(shapes.keyT, 1) <= dim(shapes.pastKeyT, 2)", | |
| "shareAppendOk": "cachedBaseOk and shareAppendShapeOk", | |
| "cachedBidirectionalBaseOk": "attrs.causal == 0 and attrs.local_window_size == -1 and metadataInputsOk and qkNormInputsOk", | |
| "shareAppendBidirectionalOk": "cachedBidirectionalBaseOk and shareAppendShapeOk", | |
| "decodeFlashShapeOk": "flashWorkgroupOk and dim(shapes.queryT, 1) == 1 and gqaHeadDim % 4 == 0 and gqaFlashHd and dim(shapes.presentKeyT, 2) >= tunables.CACHED_DECODE_MIN_KV_TOKENS and gqaDispatchFits", | |
| "qkvPrefillClusterWorkgroupSize": "qkvPrefillQueryTile * qkvPrefillLanesPerQuery", | |
| "cachedPrefillClusterWorkgroupSize": "cachedPrefillQueryTile * cachedPrefillLanesPerQuery", | |
| "cachedPrefillClusterWorkgroupOk": "cachedPrefillClusterWorkgroupSize <= deviceWorkgroupCap", | |
| "prefillFlashShapeOk": "cachedPrefillClusterWorkgroupOk and gqaHeadDim % (4 * cachedPrefillLanesPerQuery) == 0 and gqaFlashHd and dim(shapes.queryT, 1) >= cachedPrefillQueryTile and gqaDispatchFits and ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", | |
| "quantScaleShapeOk": "((attrs.k_quant_type == \"PER_TENSOR\" and numel(shapes.kScaleT) == 1) or (attrs.k_quant_type == \"PER_CHANNEL\" and numel(shapes.kScaleT) == attrs.kv_num_heads * gqaHeadDim)) and ((attrs.v_quant_type == \"PER_TENSOR\" and numel(shapes.vScaleT) == 1) or (attrs.v_quant_type == \"PER_CHANNEL\" and numel(shapes.vScaleT) == attrs.kv_num_heads * gqaHeadDim))", | |
| "quantizedNewKvOk": "cachedBaseOk and not windowCacheRequested and copyWorkgroupOk and present.kScaleT and present.vScaleT and attrs.k_quant_type == attrs.v_quant_type and attrs.k_quant_type != \"NONE\" and (attrs.kv_cache_bit_width == 4 or attrs.kv_cache_bit_width == 8) and quantScaleShapeOk and tensorDtypes.queryT == \"float32\" and tensorDtypes.keyT == \"float32\" and tensorDtypes.valueT == \"float32\" and tensorDtypes.outputT == \"float32\" and tensorDtypes.presentKeyT == (\"uint8\" if attrs.kv_cache_bit_width == 4 else \"int8\") and tensorDtypes.presentValueT == tensorDtypes.presentKeyT and dim(shapes.presentKeyT, 3) == gqaHeadDim / (8 / attrs.kv_cache_bit_width) and dim(shapes.presentValueT, 3) == dim(shapes.presentKeyT, 3) and dim(shapes.keyT, 1) > 0 and gqaHeadDim % 2 == 0", | |
| "quantizedPromptOk": "quantizedNewKvOk and dim(shapes.keyT, 1) == dim(shapes.presentKeyT, 2)", | |
| "quantizedDecodeOptionsOk": "plainAttentionOptions and not present.sinCacheT and not present.kNormWeightT", | |
| "quantizedCachedDecodeOk": "not windowCacheRequested and quantizedNewKvOk and quantizedDecodeOptionsOk and attrs.kv_cache_bit_width == 8 and present.pastKeyT and present.pastValueT and dim(shapes.queryT, 1) == 1 and dim(shapes.keyT, 1) == 1 and tensorDtypes.pastKeyT == \"int8\" and tensorDtypes.pastValueT == \"int8\" and tensorDtypes.presentKeyT == \"int8\" and tensorDtypes.presentValueT == \"int8\" and dim(shapes.pastKeyT, 0) == dim(shapes.presentKeyT, 0) and dim(shapes.pastKeyT, 1) == dim(shapes.presentKeyT, 1) and dim(shapes.pastKeyT, 2) == dim(shapes.presentKeyT, 2) and dim(shapes.pastKeyT, 3) == dim(shapes.presentKeyT, 3)", | |
| "qkvSplitScratchBytes": "dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits * headDimPlan * 4", | |
| "qkvSplitStatsBytes": "2 * dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits * 4", | |
| "qkvSplitScratchFits": "qkvSplitScratchBytes <= device.limits.maxStorageBufferBindingSize and qkvSplitScratchBytes <= device.limits.maxBufferSize and qkvSplitStatsBytes <= device.limits.maxStorageBufferBindingSize and qkvSplitStatsBytes <= device.limits.maxBufferSize", | |
| "cachedSplitScratchBytes": "dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits * gqaHeadDim * 4", | |
| "cachedSplitStatsBytes": "2 * dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits * 4", | |
| "cachedSplitScratchFits": "cachedSplitScratchBytes <= device.limits.maxStorageBufferBindingSize and cachedSplitScratchBytes <= device.limits.maxBufferSize and cachedSplitStatsBytes <= device.limits.maxStorageBufferBindingSize and cachedSplitStatsBytes <= device.limits.maxBufferSize", | |
| "qPrepScratchBytes": "numel(shapes.queryT) * 4", | |
| "qPrepScratchFits": "qPrepScratchBytes <= device.limits.maxStorageBufferBindingSize and qPrepScratchBytes <= device.limits.maxBufferSize", | |
| "qkvTiledWorkgroupBytes": "dim(shapes.valueT, 2) / attrs.kv_num_heads * 32 * 4", | |
| "qkvTiledStorageOk": "qkvTiledWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "qkvClusterTileK": "tunables.CLUSTER_TILE_K_F32 if tensorDtypes.queryT == \"float32\" and headDimPlan <= tunables.QKV_SPLIT_TILE_K else tunables.CLUSTER_TILE_K_F16", | |
| "qkvClusterWorkgroupBytes": "qkvClusterTileK * headDimPlan * (8 if tensorDtypes.queryT == \"float32\" else 4)", | |
| "qkvClusterStorageOk": "qkvClusterWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "qkvF32ClusterRegisterGeometry": "tensorDtypes.queryT != \"float32\" or (headDimPlan % 4 == 0 and headDimPlan / (4 * qkvPrefillLanesPerQuery) <= tunables.CLUSTER_MAX_SLICE)", | |
| "cachedF32ClusterRegisterGeometry": "tensorDtypes.queryT != \"float32\" or (gqaHeadDim % 4 == 0 and gqaHeadDim / (4 * cachedPrefillLanesPerQuery) <= tunables.CLUSTER_MAX_SLICE)", | |
| "qkvNoSgReductionBytesPerKey": "qkvPrefillClusterWorkgroupSize * 4", | |
| "cachedNoSgReductionBytesPerKey": "cachedPrefillClusterWorkgroupSize * 4", | |
| "qkvNoSgTileBytes": "headDimPlan * (8 if tensorDtypes.queryT == \"float32\" else 4) + qkvNoSgReductionBytesPerKey", | |
| "qkvNoSgTileK": "min(tunables.NO_SG_TILE_K_MAX, max(1, floor(device.limits.maxComputeWorkgroupStorageSize / qkvNoSgTileBytes)))", | |
| "qkvNoSgWorkgroupBytes": "qkvNoSgTileK * qkvNoSgTileBytes", | |
| "qkvNoSgClusterStorageOk": "qkvNoSgWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "qkvSgClusterOk": "qkvClusterStorageOk and qkvF32ClusterRegisterGeometry and subgroupQkvPrefillCluster", | |
| "cachedClusterTileK": "tunables.CLUSTER_TILE_K_F32 if tensorDtypes.queryT == \"float32\" and gqaHeadDim <= tunables.QKV_SPLIT_TILE_K else tunables.CLUSTER_TILE_K_F16", | |
| "cachedClusterWorkgroupBytes": "cachedClusterTileK * gqaHeadDim * (8 if tensorDtypes.queryT == \"float32\" else 4)", | |
| "cachedClusterStorageOk": "cachedClusterWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "cachedMaskClusterWorkgroupBytes": "cachedClusterWorkgroupBytes + cachedPrefillQueryTile * cachedClusterTileK * 4", | |
| "cachedMaskClusterStorageOk": "cachedMaskClusterWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "cachedNoSgTileBytes": "gqaHeadDim * (8 if tensorDtypes.queryT == \"float32\" else 4) + cachedNoSgReductionBytesPerKey", | |
| "cachedNoSgTileK": "min(tunables.NO_SG_TILE_K_MAX, max(1, floor(device.limits.maxComputeWorkgroupStorageSize / cachedNoSgTileBytes)))", | |
| "cachedNoSgWorkgroupBytes": "cachedNoSgTileK * cachedNoSgTileBytes", | |
| "cachedNoSgClusterStorageOk": "cachedNoSgWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "cachedSgPrefillOk": "cachedClusterStorageOk and subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry", | |
| "cachedMaskSgPrefillOk": "cachedMaskClusterStorageOk and subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry", | |
| "gqaMatQueryTile": "tunables.MATERIALIZED_SGMAT_QUERY_TILE", | |
| "gqaMatKeyTile": "tunables.MATERIALIZED_SGMAT_KEY_TILE", | |
| "gqaMatInnerTile": "tunables.MATERIALIZED_SGMAT_INNER_TILE", | |
| "gqaMatSubgroupRows": "floor(gqaMatQueryTile / 16)", | |
| "gqaMatSubgroupCols": "floor(gqaMatKeyTile / 32)", | |
| "gqaMatWorkgroupSize": "gqaMatSubgroupRows * gqaMatSubgroupCols * 32", | |
| "gqaMatCompactStorageBytes": "(gqaMatQueryTile + gqaMatKeyTile) * gqaMatInnerTile * 4", | |
| "gqaMatGeometryOk": "gqaMatQueryTile >= 16 and gqaMatQueryTile % 16 == 0 and gqaMatKeyTile >= 32 and gqaMatKeyTile <= 64 and gqaMatKeyTile % 32 == 0 and gqaMatInnerTile == 32 and gqaMatQueryTile >= 64 and gqaMatKeyTile >= 64", | |
| "gqaMatResourcesFit": "gqaMatGeometryOk and gqaMatWorkgroupSize <= deviceWorkgroupCap and gqaMatCompactStorageBytes <= device.limits.maxComputeWorkgroupStorageSize", | |
| "gqaMatScoreBytes": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.keyT, 1) * 4", | |
| "gqaMatScoreFits": "gqaMatScoreBytes <= device.limits.maxStorageBufferBindingSize and gqaMatScoreBytes <= device.limits.maxBufferSize", | |
| "gqaMatBuffersFit": "numel(shapes.queryT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.keyT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.valueT) * 4 <= device.limits.maxStorageBufferBindingSize", | |
| "gqaMatDispatchFits": "ceilDiv(dim(shapes.keyT, 1), gqaMatKeyTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and dim(shapes.queryT, 0) * attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", | |
| "gqaMatStatSlots": "ceilDiv(dim(shapes.keyT, 1), gqaMatKeyTile) * gqaMatSubgroupCols", | |
| "gqaMatRowStatsWg": "min(tunables.MATERIALIZED_SOFTMAX_WORKGROUP_SIZE, deviceWorkgroupCap)", | |
| "gqaMatRowStatsElements": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * 2", | |
| "gqaMatScorePartialElements": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatStatSlots * dim(shapes.queryT, 1) * 2", | |
| "gqaMatCoreOk": "qkvContractOk and attrs.causal == 0 and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and headDimPlan >= 64 and headDimPlan <= 256 and headDimPlan % 16 == 0 and dim(shapes.queryT, 1) >= tunables.MATERIALIZED_SGMAT_MIN_SEQ and dim(shapes.keyT, 1) >= tunables.MATERIALIZED_SGMAT_MIN_SEQ and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatScoreFits and gqaMatBuffersFit and gqaMatResourcesFit and gqaMatDispatchFits", | |
| "gqaMatPastQSeq": "dim(shapes.queryT, 1)", | |
| "gqaMatPastKvSeq": "dim(shapes.presentKeyT, 2)", | |
| "gqaMatPastScoreBytes": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatPastQSeq * gqaMatPastKvSeq * 4", | |
| "gqaMatPastScoreFits": "gqaMatPastScoreBytes <= device.limits.maxStorageBufferBindingSize and gqaMatPastScoreBytes <= device.limits.maxBufferSize", | |
| "gqaMatPastBuffersFit": "numel(shapes.queryT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.presentKeyT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.presentValueT) * 4 <= device.limits.maxStorageBufferBindingSize", | |
| "gqaMatPastStatSlots": "ceilDiv(gqaMatPastKvSeq, gqaMatKeyTile) * gqaMatSubgroupCols", | |
| "gqaMatPastRowStatsElements": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatPastQSeq * 2", | |
| "gqaMatPastScorePartialElements": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatPastStatSlots * gqaMatPastQSeq * 2", | |
| "gqaMatPastDispatchFits": "ceilDiv(gqaMatPastKvSeq, gqaMatKeyTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and ceilDiv(gqaMatPastQSeq, gqaMatQueryTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and dim(shapes.queryT, 0) * attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", | |
| "gqaMatShareAppendOk": "shareAppendOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and dim(shapes.queryT, 1) == dim(shapes.keyT, 1) and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and dim(shapes.queryT, 1) >= tunables.MATERIALIZED_SGMAT_APPEND_MIN_QSEQ and dim(shapes.presentKeyT, 2) >= dim(shapes.queryT, 1) and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "gqaMatWindowOk": "windowShiftOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size > 0 and dim(shapes.queryT, 1) == dim(shapes.keyT, 1) and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and dim(shapes.queryT, 1) >= tunables.MATERIALIZED_SGMAT_WINDOW_MIN_QSEQ and windowCapacity >= dim(shapes.queryT, 1) and attrs.local_window_size + dim(shapes.queryT, 1) >= windowCapacity and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "gqaMatPastCoreOk": "sharedKvFloatOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "gqaMatPastRotaryOk": "sharedKvFloatOk and standardSoftmax and present.cosCacheT and rotaryRequested and not present.qNormWeightT and not present.attentionBiasT and not present.headSinkT and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaHeadDim % 8 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "gqaMatNewPastOk": "newKvPastOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "gqaMatPastCoreF16Ok": "sharedKvFloatOk and plainAttentionOptions and tensorDtypes.queryT == \"float16\" and device.features.has(\"shader-f16\") and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "gqaMatPastRotaryF16Ok": "sharedKvFloatOk and standardSoftmax and present.cosCacheT and rotaryRequested and not present.qNormWeightT and not present.attentionBiasT and not present.headSinkT and tensorDtypes.queryT == \"float16\" and device.features.has(\"shader-f16\") and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaHeadDim % 8 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "gqaMatPastBiasOk": "sharedKvFloatOk and standardSoftmax and not present.cosCacheT and not rotaryRequested and not present.qNormWeightT and not present.headSinkT and present.attentionBiasT and ranks.attentionBiasT == 4 and tensorDtypes.attentionBiasT == \"float32\" and (dim(shapes.attentionBiasT, 0) == dim(shapes.queryT, 0) or dim(shapes.attentionBiasT, 0) == 1) and (dim(shapes.attentionBiasT, 1) == attrs.num_heads or dim(shapes.attentionBiasT, 1) == 1) and dim(shapes.attentionBiasT, 2) == dim(shapes.queryT, 1) and dim(shapes.attentionBiasT, 3) == dim(shapes.presentKeyT, 2) and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", | |
| "qNumHeads": "attrs.num_heads", | |
| "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", | |
| "materializedSgmatQueryTile": "gqaMatQueryTile", | |
| "materializedSgmatKeyTile": "gqaMatKeyTile", | |
| "materializedSgmatInnerTile": "gqaMatInnerTile", | |
| "materializedSgmatRuntimeDirectStore": false, | |
| "materializedRowStatsWg": "gqaMatRowStatsWg", | |
| "statQuerySeq": "dim(shapes.queryT, 1)", | |
| "hasWindow": "attrs.local_window_size > 0" | |
| }, | |
| "bindings": { | |
| "query": { "arg": "queryT", "elementType": "f32" }, | |
| "key": { "arg": "keyT", "elementType": "f32" }, | |
| "scores": { "scratch": "materializedScores", "elementType": "f32" }, | |
| "scorePartials": { "scratch": "materializedScorePartials", "elementType": "f32" }, | |
| "scorePartials_f32": { | |
| "scratch": "materializedScorePartials", | |
| "name": "scorePartials", | |
| "buffer": "read-only-storage", | |
| "elementType": "f32" | |
| }, | |
| "rowStats": { "scratch": "materializedRowStats", "elementType": "f32" }, | |
| "scores_f32": { | |
| "scratch": "materializedScores", | |
| "name": "scores", | |
| "buffer": "read-only-storage", | |
| "elementType": "f32" | |
| }, | |
| "value": { "arg": "valueT", "elementType": "f32" }, | |
| "rowStats_f32": { | |
| "scratch": "materializedRowStats", | |
| "name": "rowStats", | |
| "buffer": "read-only-storage", | |
| "elementType": "f32" | |
| }, | |
| "output": { "arg": "outputT", "elementType": "f32" }, | |
| "key_key_t": { "arg": "keyT", "name": "key", "elementType": "$inputElement" }, | |
| "value_value_t": { "arg": "valueT", "name": "value", "elementType": "$inputElement" }, | |
| "present_key": { "arg": "presentKeyT", "elementType": "$presentElement" }, | |
| "present_value": { "arg": "presentValueT", "elementType": "$presentElement" }, | |
| "params": { | |
| "struct": [ | |
| { "name": "batchSize", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } | |
| ] | |
| }, | |
| "src_k": { "arg": "pastKeyT", "elementType": "$inputScalar" }, | |
| "src_v": { "arg": "pastValueT", "elementType": "$inputScalar" }, | |
| "present_key_present": { "arg": "presentKeyT", "name": "present_key", "elementType": "$inputScalar" }, | |
| "present_value_present": { "arg": "presentValueT", "name": "present_value", "elementType": "$inputScalar" }, | |
| "params_count": { | |
| "name": "params", | |
| "struct": [ | |
| { | |
| "name": "count", | |
| "type": "u32", | |
| "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)" | |
| } | |
| ] | |
| }, | |
| "key_f32": { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "f32" }, | |
| "params_scores": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } | |
| ] | |
| }, | |
| "params_rows": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "rows", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)" } | |
| ] | |
| }, | |
| "value_f32": { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "f32" }, | |
| "params_apply": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" } | |
| ] | |
| }, | |
| "key_f16": { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "f16" }, | |
| "value_f16": { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "f16" }, | |
| "output_f16": { "arg": "outputT", "name": "output", "elementType": "f16" }, | |
| "query_qprep": { "arg": "queryT", "name": "query", "elementType": "$inputScalar" }, | |
| "cos_cache": { "arg": "cosCacheT", "elementType": "$cosScalar" }, | |
| "sin_cache": { "arg": "sinCacheT", "elementType": "$cosScalar" }, | |
| "qout": { "scratch": "qPrep", "elementType": "vec4<f32>" }, | |
| "params_qprep": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" } | |
| ] | |
| }, | |
| "past_k": { "arg": "pastKeyT", "elementType": "$inputScalar" }, | |
| "past_v": { "arg": "pastValueT", "elementType": "$inputScalar" }, | |
| "new_k": { "arg": "keyT", "elementType": "$inputScalar" }, | |
| "new_v": { "arg": "valueT", "elementType": "$inputScalar" }, | |
| "params_present": { | |
| "name": "params", | |
| "struct": [ | |
| { | |
| "name": "count", | |
| "type": "u32", | |
| "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)" | |
| }, | |
| { "name": "seq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "pastSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { "name": "keySeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } | |
| ] | |
| }, | |
| "seqlens_k": { "arg": "seqlensKT", "elementType": "i32" }, | |
| "params__uniform": { | |
| "name": "params", | |
| "struct": [ | |
| { | |
| "name": "count", | |
| "type": "u32", | |
| "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)" | |
| }, | |
| { "name": "seq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "keySeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } | |
| ] | |
| }, | |
| "params_gqa_present": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, | |
| { "name": "seq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "keySeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } | |
| ] | |
| }, | |
| "kcache": { "arg": "presentKeyT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, | |
| "vcache": { "arg": "presentValueT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, | |
| "output_past_kv": { "arg": "outputT", "name": "output", "elementType": "$inputScalar" }, | |
| "params_attention": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, | |
| { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" } | |
| ] | |
| }, | |
| "head_sink": { "arg": "headSinkT", "elementType": "$inputScalar" }, | |
| "cos_cache_f32": { "arg": "cosCacheT", "name": "cos_cache", "elementType": "f32" }, | |
| "sin_cache_f32": { "arg": "sinCacheT", "name": "sin_cache", "elementType": "f32" }, | |
| "query_query_t": { "arg": "queryT", "name": "query", "elementType": "$inputElement" }, | |
| "output_output_t": { "arg": "outputT", "name": "output", "elementType": "$outputElement" }, | |
| "params_qkv_present": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| } | |
| ] | |
| }, | |
| "params_qkv_present_causal": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "isCausal", "type": "u32", "value": "attrs.causal" } | |
| ] | |
| }, | |
| "src_k_f32": { "arg": "keyT", "name": "src_k", "elementType": "f32" }, | |
| "src_v_f32": { "arg": "valueT", "name": "src_v", "elementType": "f32" }, | |
| "k_scale": { "arg": "kScaleT", "elementType": "f32" }, | |
| "v_scale": { "arg": "vScaleT", "elementType": "f32" }, | |
| "present_key_quant_int8": { "arg": "presentKeyT", "name": "present_key", "elementType": "$quantCacheElement" }, | |
| "present_value_quant_int8": { "arg": "presentValueT", "name": "present_value", "elementType": "$quantCacheElement" }, | |
| "params_quant_int8": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, | |
| { "name": "seq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } | |
| ] | |
| }, | |
| "partial_out": { "scratch": "partialOut", "elementType": "vec4<f32>" }, | |
| "partial_stats": { "scratch": "partialStats", "elementType": "vec2<f32>" }, | |
| "partial_out_merge": { | |
| "scratch": "partialOut", | |
| "name": "partial_out", | |
| "buffer": "read-only-storage", | |
| "elementType": "vec4<f32>" | |
| }, | |
| "partial_stats_merge": { | |
| "scratch": "partialStats", | |
| "name": "partial_stats", | |
| "buffer": "read-only-storage", | |
| "elementType": "vec2<f32>" | |
| }, | |
| "output_merge": { "arg": "outputT", "name": "output", "elementType": "$inputVec4" }, | |
| "query_attention": { "arg": "queryT", "name": "query", "elementType": "$inputVec4" }, | |
| "key_attention": { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "$inputVec4" }, | |
| "value_attention": { | |
| "arg": "presentValueT", | |
| "name": "value", | |
| "buffer": "read-only-storage", | |
| "elementType": "$inputVec4" | |
| }, | |
| "attn_mask": { "arg": "attentionBiasT", "elementType": "$inputScalar" }, | |
| "params_attn_flash_prefill_cluster": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "isCausal", "type": "u32", "value": 1 }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { | |
| "name": "maskBatchStride", | |
| "type": "u32", | |
| "value": "0 if dim(shapes.attentionBiasT, 0) == 1 else dim(shapes.attentionBiasT, 1) * dim(shapes.attentionBiasT, 2) * dim(shapes.attentionBiasT, 3)" | |
| }, | |
| { | |
| "name": "maskHeadStride", | |
| "type": "u32", | |
| "value": "0 if dim(shapes.attentionBiasT, 1) == 1 else dim(shapes.attentionBiasT, 2) * dim(shapes.attentionBiasT, 3)" | |
| }, | |
| { "name": "maskSeqStride", "type": "u32", "value": "dim(shapes.attentionBiasT, 3)" } | |
| ] | |
| }, | |
| "q_norm_weight": { "arg": "qNormWeightT", "elementType": "$qNormScalar" }, | |
| "params_past_kv_flash_prefill": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "isCausal", "type": "u32", "value": 1 }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| } | |
| ] | |
| }, | |
| "params_split_attention": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| } | |
| ] | |
| }, | |
| "params_new_kv_past_flash_prefill": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "isCausal", "type": "u32", "value": 1 }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| } | |
| ] | |
| }, | |
| "params_new_kv_past_flash_bidirectional": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" } | |
| ] | |
| }, | |
| "params_past_kv_softcap_flash_prefill": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "isCausal", "type": "u32", "value": 1 }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" } | |
| ] | |
| }, | |
| "kcache_past_kv": { "arg": "pastKeyT", "name": "kcache", "elementType": "$inputScalar" }, | |
| "vcache_past_kv": { "arg": "pastValueT", "name": "vcache", "elementType": "$inputScalar" }, | |
| "params_past_kv": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, | |
| { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" } | |
| ] | |
| }, | |
| "k_norm_weight": { "arg": "kNormWeightT", "elementType": "f32" }, | |
| "present_key_f32": { "arg": "presentKeyT", "name": "present_key", "elementType": "f32" }, | |
| "present_value_f32": { "arg": "presentValueT", "name": "present_value", "elementType": "f32" }, | |
| "params_count_seq_past_seq": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, | |
| { "name": "seq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "pastSeq", "type": "u32", "value": 0 } | |
| ] | |
| }, | |
| "kcache_f32": { "arg": "presentKeyT", "name": "kcache", "buffer": "read-only-storage", "elementType": "f32" }, | |
| "vcache_f32": { "arg": "presentValueT", "name": "vcache", "buffer": "read-only-storage", "elementType": "f32" }, | |
| "q_norm_weight_f32": { "arg": "qNormWeightT", "name": "q_norm_weight", "elementType": "f32" }, | |
| "kcache_f32_read": { "arg": "pastKeyT", "name": "kcache", "elementType": "f32" }, | |
| "vcache_f32_read": { "arg": "pastValueT", "name": "vcache", "elementType": "f32" }, | |
| "attn_bias": { "arg": "attentionBiasT", "elementType": "$inputScalar" }, | |
| "params_past_kv_bias": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, | |
| { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" }, | |
| { "name": "biasBatch", "type": "u32", "value": "dim(shapes.attentionBiasT, 0)" }, | |
| { "name": "biasHeads", "type": "u32", "value": "dim(shapes.attentionBiasT, 1)" } | |
| ] | |
| }, | |
| "params_window_bias": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, | |
| { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" }, | |
| { "name": "biasBatch", "type": "u32", "value": "dim(shapes.attentionBiasT, 0)" }, | |
| { "name": "biasHeads", "type": "u32", "value": "dim(shapes.attentionBiasT, 1)" }, | |
| { "name": "biasCols", "type": "u32", "value": "dim(shapes.attentionBiasT, 3)" } | |
| ] | |
| }, | |
| "params_gqa_attention": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" } | |
| ] | |
| }, | |
| "params_past_kv_bias_headsink": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, | |
| { "name": "biasBatch", "type": "u32", "value": "dim(shapes.attentionBiasT, 0)" }, | |
| { "name": "biasHeads", "type": "u32", "value": "dim(shapes.attentionBiasT, 1)" } | |
| ] | |
| }, | |
| "key_present_key_t": { | |
| "arg": "presentKeyT", | |
| "name": "key", | |
| "buffer": "read-only-storage", | |
| "elementType": "$quantCacheReadElement" | |
| }, | |
| "value_present_value_t": { | |
| "arg": "presentValueT", | |
| "name": "value", | |
| "buffer": "read-only-storage", | |
| "elementType": "$quantCacheReadElement" | |
| }, | |
| "params_quant_int8_flash_prefill": { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "isCausal", "type": "u32", "value": 1 }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } | |
| ] | |
| } | |
| }, | |
| "variants": [ | |
| { | |
| "id": "qkv_present_materialized_sgmat_f32", | |
| "priority": 23, | |
| "when": ["gqaMatCoreOk"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "statSlots": "gqaMatStatSlots", | |
| "presentScalar": "qkvCacheScalar", | |
| "presentElement": "qkvCacheVec4", | |
| "presentVec4": true, | |
| "inputElement": "qkvInputVec4", | |
| "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", | |
| "kvHiddenV4": "dim(shapes.keyT, 2) / 4" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.keyT, 1)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.MaterializedScoresSgmat", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"" }, | |
| "bindings": [ | |
| "query", | |
| "key", | |
| "scores", | |
| "scorePartials", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if has(attrs, \"scale\") else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.keyT, 1), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.MaterializedRowStatsCombine", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": [ | |
| "scorePartials_f32", | |
| "rowStats", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { | |
| "name": "rows", | |
| "type": "u32", | |
| "value": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)" | |
| } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.MaterializedApplySgmat", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"" }, | |
| "bindings": [ | |
| "scores_f32", | |
| "value", | |
| "rowStats_f32", | |
| "output", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(headDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.MaterializedPresent", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_materialized_sgmat_f32", | |
| "priority": 35, | |
| "when": ["gqaMatPastCoreOk"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"copy\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.PastMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["query", "key_f32", "scores", "scorePartials", "params_scores"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.PastMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.PastMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["scores_f32", "value_f32", "rowStats_f32", "output", "params_apply"], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_bias_materialized_sgmat_f32", | |
| "priority": 36, | |
| "when": ["gqaMatPastBiasOk"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"copy\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.PastMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true, | |
| "scoreBias": true | |
| }, | |
| "bindings": [ | |
| "query", | |
| "key_f32", | |
| { "arg": "attentionBiasT", "name": "attention_bias", "elementType": "f32" }, | |
| "scores", | |
| "scorePartials", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "biasBatch", "type": "u32", "value": "dim(shapes.attentionBiasT, 0)" }, | |
| { "name": "biasHeads", "type": "u32", "value": "dim(shapes.attentionBiasT, 1)" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.PastMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.PastMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["scores_f32", "value_f32", "rowStats_f32", "output", "params_apply"], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_materialized_sgmat_f16", | |
| "priority": 35, | |
| "when": ["gqaMatPastCoreF16Ok"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f16", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"copy\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar", | |
| "operandF16": true | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.PastMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": [ | |
| { "arg": "queryT", "name": "query", "elementType": "f16" }, | |
| "key_f16", | |
| "scores", | |
| "scorePartials", | |
| "params_scores" | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.PastMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.PastMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["scores_f32", "value_f16", "rowStats_f32", "output_f16", "params_apply"], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_rotary_materialized_sgmat_f32", | |
| "priority": 37, | |
| "when": ["gqaMatPastRotaryOk"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"copy\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar", | |
| "half": "gqaHeadDim / 2", | |
| "qHeads": "attrs.num_heads", | |
| "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", | |
| "hasQNorm": false | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "qPrep", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0), dim(shapes.queryT, 1), dim(shapes.queryT, 2)]" | |
| }, | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "qprep", | |
| "name": "GroupQueryAttention.RotaryQPrep", | |
| "shader": "gqa-qprep.wgsl.jinja", | |
| "bindings": ["query_qprep", "cos_cache", "sin_cache", "qout", "params_qprep"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.PastRotaryMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": [ | |
| { "scratch": "qPrep", "name": "query", "buffer": "read-only-storage", "elementType": "f32" }, | |
| "key_f32", | |
| "scores", | |
| "scorePartials", | |
| "params_scores" | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.PastRotaryMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.PastRotaryMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["scores_f32", "value_f32", "rowStats_f32", "output", "params_apply"], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_rotary_materialized_sgmat_f16", | |
| "priority": 37, | |
| "when": ["gqaMatPastRotaryF16Ok"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f16", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"copy\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar", | |
| "half": "gqaHeadDim / 2", | |
| "qHeads": "attrs.num_heads", | |
| "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", | |
| "hasQNorm": false, | |
| "operandF16": true | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "qPrep", | |
| "dtype": "float16", | |
| "shape": "[dim(shapes.queryT, 0), dim(shapes.queryT, 1), dim(shapes.queryT, 2)]" | |
| }, | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "qprep", | |
| "name": "GroupQueryAttention.RotaryQPrep", | |
| "shader": "gqa-qprep.wgsl.jinja", | |
| "bindings": [ | |
| "query_qprep", | |
| "cos_cache", | |
| "sin_cache", | |
| { "scratch": "qPrep", "name": "qout", "elementType": "vec4<f16>" }, | |
| "params_qprep" | |
| ], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "derive": { "qPrepF16": true } | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.PastRotaryMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": [ | |
| { "scratch": "qPrep", "name": "query", "buffer": "read-only-storage", "elementType": "f16" }, | |
| "key_f16", | |
| "scores", | |
| "scorePartials", | |
| "params_scores" | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.PastRotaryMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.PastRotaryMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["scores_f32", "value_f16", "rowStats_f32", "output_f16", "params_apply"], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_past_materialized_sgmat_f32", | |
| "priority": 35, | |
| "when": ["gqaMatNewPastOk"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"merge\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Merge", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "params_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.NewPastMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["query", "key_f32", "scores", "scorePartials", "params_scores"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.NewPastMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.NewPastMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true | |
| }, | |
| "bindings": ["scores_f32", "value_f32", "rowStats_f32", "output", "params_apply"], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_materialized_sgmat_f32", | |
| "priority": 36, | |
| "when": ["gqaMatWindowOk"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"window_shift\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.WindowMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true, | |
| "scoreWindow": true, | |
| "useSeqlens": true | |
| }, | |
| "bindings": [ | |
| "query", | |
| "key_f32", | |
| "scores", | |
| "scorePartials", | |
| "seqlens_k", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.WindowMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.WindowMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true, | |
| "scoreWindow": true, | |
| "useSeqlens": true | |
| }, | |
| "bindings": [ | |
| "scores_f32", | |
| "value_f32", | |
| "rowStats_f32", | |
| "output", | |
| "seqlens_k", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "share_append_materialized_sgmat_f32", | |
| "priority": 36, | |
| "when": ["gqaMatShareAppendOk"], | |
| "requires": { | |
| "features": ["subgroups", "chromium-experimental-subgroup-matrix"], | |
| "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] | |
| }, | |
| "derive": { | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "statSlots": "gqaMatPastStatSlots", | |
| "mode": "\"merge_share\"", | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputScalar": "gqaScalar", | |
| "useSeqlens": true | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "materializedScores", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" | |
| }, | |
| { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, | |
| { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "scores", | |
| "name": "GroupQueryAttention.ShareAppendMaterializedScores", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"score\"", | |
| "emitRowStats": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true, | |
| "useSeqlens": true | |
| }, | |
| "bindings": ["query", "key_f32", "scores", "scorePartials", "seqlens_k", "params_scores"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| }, | |
| { | |
| "id": "rowstats", | |
| "name": "GroupQueryAttention.ShareAppendMaterializedRowStats", | |
| "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", | |
| "bindings": ["scorePartials_f32", "rowStats", "params_rows"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": 1, | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "apply", | |
| "name": "GroupQueryAttention.ShareAppendMaterializedApply", | |
| "shader": "attn-materialized-sgmat-f32.wgsl.jinja", | |
| "derive": { | |
| "phase": "\"apply\"", | |
| "fusedSoftmax": true, | |
| "layout": "\"bsh\"", | |
| "kvLayout": "\"bhsd\"", | |
| "causalRightAlign": true, | |
| "useSeqlens": true | |
| }, | |
| "bindings": ["scores_f32", "value_f32", "rowStats_f32", "output", "seqlens_k", "params_apply"], | |
| "dispatch": { | |
| "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", | |
| "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", | |
| "z": "dim(shapes.queryT, 0) * attrs.num_heads" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_share_append_split", | |
| "priority": 32, | |
| "when": ["shareAppendOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "not present.kNormWeightT", "gqaHeadDim % 2 == 0"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge_share\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache", "vcache", "output_past_kv", "seqlens_k", "params_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_share_append_headsink_split", | |
| "priority": 30, | |
| "when": ["shareAppendOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "not present.kNormWeightT", "not present.attentionBiasT", "present.headSinkT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge_share\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar", | |
| "hasHeadSink": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": [ | |
| "query_qprep", | |
| "kcache", | |
| "vcache", | |
| "head_sink", | |
| "output_past_kv", | |
| "seqlens_k", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_share_append_rotary_split", | |
| "priority": 30, | |
| "when": ["shareAppendOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "present.cosCacheT and rotaryRequested", "not present.qNormWeightT", "not present.kNormWeightT", "not present.attentionBiasT", "not present.headSinkT", "tensorDtypes.queryT == \"float32\""], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge_share\"", | |
| "inputScalar": "\"f32\"", | |
| "half": "gqaHeadDim / 2", | |
| "hasRotary": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "cos_cache_f32", "sin_cache_f32", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache", "vcache", "cos_cache_f32", "sin_cache_f32", "output_past_kv", "seqlens_k", "params_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_share_append_bidirectional_split", | |
| "priority": 32, | |
| "when": ["shareAppendBidirectionalOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "not present.kNormWeightT", "gqaHeadDim % 2 == 0"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge_share\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar", | |
| "bidirectional": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache", "vcache", "output_past_kv", "seqlens_k", "params_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_tiled_nosg", | |
| "priority": 19, | |
| "when": ["plainAttentionOptions", "flashShapeOk", "qkvTiledStorageOk", "headDimPlan % 4 == 0", "headDimPlan <= 128", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "attrs.local_window_size <= 0", "not causalNoWindow"], | |
| "derive": { | |
| "scalar": "qkvScalar", | |
| "inputScalar": "qkvScalar", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "qkvScalar", | |
| "presentElement": "qkvCacheScalar", | |
| "blockM": 32, | |
| "vHeadCap": "dim(shapes.valueT, 2) / attrs.kv_num_heads", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "kvHidden": "dim(shapes.keyT, 2)" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.PrefillTiledNoSg", | |
| "shader": "attention-rank4-tiled.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"" }, | |
| "bindings": [ | |
| { "arg": "queryT", "name": "q", "elementType": "$scalar" }, | |
| { "arg": "keyT", "name": "k", "elementType": "$scalar" }, | |
| { "arg": "valueT", "name": "v", "elementType": "$scalar" }, | |
| { "arg": "outputT", "name": "y", "elementType": "$scalar" }, | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "count", "type": "u32", "value": "numel(shapes.outputT)" }, | |
| { "name": "qHeads", "type": "u32", "value": "attrs.num_heads" }, | |
| { "name": "kvHeads", "type": "u32", "value": "attrs.kv_num_heads" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "headSize", "type": "u32", "value": "dim(shapes.queryT, 2) / attrs.num_heads" }, | |
| { "name": "vHeadSize", "type": "u32", "value": "dim(shapes.valueT, 2) / attrs.kv_num_heads" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "0" }, | |
| { "name": "isCausal", "type": "u32", "value": "0" }, | |
| { "name": "qHidden", "type": "u32", "value": "dim(shapes.queryT, 2)" }, | |
| { "name": "kvHidden", "type": "u32", "value": "dim(shapes.keyT, 2)" }, | |
| { "name": "vHidden", "type": "u32", "value": "dim(shapes.valueT, 2)" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.outputT, 0) * attrs.num_heads * ceilDiv(dim(shapes.outputT, 1), blockM) * blockM), (blockM)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.outputT, 0) * attrs.num_heads * ceilDiv(dim(shapes.outputT, 1), blockM) * blockM), (blockM)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.PresentTiledNoSg", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_flash", | |
| "priority": 20, | |
| "when": ["flashShapeOk", "not causalNoWindow"], | |
| "requires": { "features": ["subgroups"] }, | |
| "derive": { | |
| "hasCausal": false, | |
| "combineSubgroups": true, | |
| "scalar": "qkvScalar", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "qkvInputVec4", | |
| "outputElement": "qkvInputVec4", | |
| "presentElement": "qkvCacheVec4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "kvHiddenV4": "dim(shapes.keyT, 2) / 4", | |
| "presentVec4": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Flash", | |
| "shader": "attn-flash-online.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"" }, | |
| "bindings": ["query_query_t", "key_key_t", "value_value_t", "output_output_t", "params_qkv_present"], | |
| "dispatch": { "x": "dim(shapes.queryT, 1)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_flash_causal", | |
| "priority": 20, | |
| "when": ["flashShapeOk", "causalNoWindow", "dim(shapes.keyT, 1) == dim(shapes.queryT, 1)"], | |
| "requires": { "features": ["subgroups"] }, | |
| "derive": { | |
| "hasCausal": true, | |
| "combineSubgroups": true, | |
| "scalar": "qkvScalar", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "qkvInputVec4", | |
| "outputElement": "qkvInputVec4", | |
| "presentElement": "qkvCacheVec4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "kvHiddenV4": "dim(shapes.keyT, 2) / 4", | |
| "presentVec4": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Flash", | |
| "shader": "attn-flash-online.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"" }, | |
| "bindings": ["query_query_t", "key_key_t", "value_value_t", "output_output_t", "params_qkv_present_causal"], | |
| "dispatch": { "x": "dim(shapes.queryT, 1)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present", | |
| "priority": 0, | |
| "when": ["fallbackShapeOk", "not causalNoWindow"], | |
| "derive": { | |
| "hasCausal": false, | |
| "hasKeyLimit": false, | |
| "inputScalar": "qkvScalar", | |
| "outputScalar": "qkvScalar", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "qkvScalar", | |
| "outputElement": "qkvScalar", | |
| "presentElement": "qkvCacheScalar", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "workgroupSize": "min(256, pow2ceil(dim(shapes.queryT, 2) / attrs.num_heads))" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attention", | |
| "shader": "attn-online-scalar.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"" }, | |
| "bindings": ["query_query_t", "key_key_t", "value_value_t", "output_output_t", "params_qkv_present"], | |
| "dispatch": { | |
| "x": "min(dim(shapes.queryT, 1), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_causal", | |
| "priority": 0, | |
| "when": ["fallbackShapeOk", "causalNoWindow", "dim(shapes.keyT, 1) == dim(shapes.queryT, 1)"], | |
| "derive": { | |
| "hasCausal": true, | |
| "hasKeyLimit": false, | |
| "inputScalar": "qkvScalar", | |
| "outputScalar": "qkvScalar", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "qkvScalar", | |
| "outputElement": "qkvScalar", | |
| "presentElement": "qkvCacheScalar", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "workgroupSize": "min(256, pow2ceil(dim(shapes.queryT, 2) / attrs.num_heads))" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attention", | |
| "shader": "attn-online-scalar.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"" }, | |
| "bindings": ["query_query_t", "key_key_t", "value_value_t", "output_output_t", "params_qkv_present_causal"], | |
| "dispatch": { | |
| "x": "min(dim(shapes.queryT, 1), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "quant_int8", | |
| "priority": 35, | |
| "when": ["quantizedPromptOk", "gqaScalarDispatchOk", "attrs.kv_cache_bit_width == 8"], | |
| "derive": { | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim / (8 / 8)", | |
| "quantized": true, | |
| "bits": 8, | |
| "mode": "\"build_quant\"", | |
| "qmax": "127 if 8 == 8 else 7", | |
| "qmin": "0 - (128 if 8 == 8 else 8)", | |
| "quantCacheElement": "\"i32\" if 8 == 8 else \"u32\"" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k_f32", "src_v_f32", "k_scale", "v_scale", "present_key_quant_int8", "present_value_quant_int8", "params_quant_int8"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": [ | |
| "query", | |
| { | |
| "arg": "presentKeyT", | |
| "name": "kcache", | |
| "buffer": "read-only-storage", | |
| "elementType": "$quantCacheElement" | |
| }, | |
| { | |
| "arg": "presentValueT", | |
| "name": "vcache", | |
| "buffer": "read-only-storage", | |
| "elementType": "$quantCacheElement" | |
| }, | |
| "k_scale", | |
| "v_scale", | |
| "output", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, | |
| { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" }, | |
| { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "quant_int4", | |
| "priority": 35, | |
| "when": ["quantizedPromptOk", "gqaScalarDispatchOk", "attrs.kv_cache_bit_width == 4"], | |
| "derive": { | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim / (8 / 4)", | |
| "quantized": true, | |
| "bits": 4, | |
| "mode": "\"build_quant\"", | |
| "qmax": "127 if 4 == 8 else 7", | |
| "qmin": "0 - (128 if 4 == 8 else 8)", | |
| "quantCacheElement": "\"i32\" if 4 == 8 else \"u32\"" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k_f32", "src_v_f32", "k_scale", "v_scale", "present_key_quant_int8", "present_value_quant_int8", "params_quant_int8"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": [ | |
| "query", | |
| { | |
| "arg": "presentKeyT", | |
| "name": "kcache", | |
| "buffer": "read-only-storage", | |
| "elementType": "$quantCacheElement" | |
| }, | |
| { | |
| "arg": "presentValueT", | |
| "name": "vcache", | |
| "buffer": "read-only-storage", | |
| "elementType": "$quantCacheElement" | |
| }, | |
| "k_scale", | |
| "v_scale", | |
| "output", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, | |
| { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" }, | |
| { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "quant_int8_decode_splitk", | |
| "priority": 45, | |
| "when": ["quantizedCachedDecodeOk", "decodeFlashShapeOk", "cachedNumSplits >= 2", "cachedSplitScratchFits"], | |
| "derive": { | |
| "useSubgroups": "subgroupSplitK", | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "numSplits": "cachedNumSplits", | |
| "usesF16": false, | |
| "scalar": "\"f32\"", | |
| "inputScalar": "\"i32\"", | |
| "inputVec4": "\"vec4<f32>\"", | |
| "quantizedCache": true, | |
| "cacheSeqlens": true, | |
| "bits": 8, | |
| "packed": "gqaHeadDim", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qmax": 127, | |
| "qmin": -128 | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "partialOut", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits * gqaHeadDim]" | |
| }, | |
| { | |
| "id": "partialStats", | |
| "dtype": "float32", | |
| "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits]" | |
| } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "copy_past", | |
| "name": "GroupQueryAttention.QuantInt8CopyPast", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "modeSpec": "\"copy\"" }, | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "append", | |
| "name": "GroupQueryAttention.QuantInt8Append", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "modeSpec": "\"append_quant\"" }, | |
| "bindings": [ | |
| "src_k_f32", | |
| "src_v_f32", | |
| "k_scale", | |
| "v_scale", | |
| "seqlens_k", | |
| { "arg": "presentKeyT", "name": "present_key", "elementType": "i32" }, | |
| { "arg": "presentValueT", "name": "present_value", "elementType": "i32" }, | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { | |
| "name": "count", | |
| "type": "u32", | |
| "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" | |
| }, | |
| { "name": "seq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "split_attention", | |
| "name": "GroupQueryAttention.QuantInt8DecodeSplitK", | |
| "shader": "attn-flash-decode-splitk.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": [ | |
| { "arg": "queryT", "name": "query", "elementType": "vec4<f32>" }, | |
| { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "vec4<i32>" }, | |
| { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "vec4<i32>" }, | |
| "k_scale", | |
| "v_scale", | |
| "seqlens_k", | |
| "partial_out", | |
| "partial_stats", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| }, | |
| { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "merge", | |
| "name": "GroupQueryAttention.QuantInt8DecodeSplitKMerge", | |
| "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["partial_out_merge", "partial_stats_merge", "output_merge"], | |
| "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_flash_splitk", | |
| "priority": 25, | |
| "when": ["decodeSplitKShapeOk", "qkvSplitScratchFits", "not causalNoWindow"], | |
| "derive": { | |
| "useSubgroups": "device.features.has(\"subgroups\")", | |
| "scalar": "qkvScalar", | |
| "inputVec4": "qkvInputVec4", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "qkvInputVec4", | |
| "presentElement": "qkvCacheVec4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "kvHiddenV4": "dim(shapes.keyT, 2) / 4", | |
| "numSplits": "qkvNumSplits", | |
| "presentVec4": true, | |
| "usesF16": "qkvUsesF16" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "partialOut", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits * (dim(shapes.queryT, 2) / attrs.num_heads)]" | |
| }, | |
| { | |
| "id": "partialStats", | |
| "dtype": "float32", | |
| "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits]" | |
| } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "split_attention", | |
| "name": "GroupQueryAttention.DecodeSplitK", | |
| "shader": "attn-flash-decode-splitk.wgsl.jinja", | |
| "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, | |
| "bindings": [ | |
| "query_attention", | |
| { "arg": "keyT", "name": "key", "elementType": "$inputVec4" }, | |
| { "arg": "valueT", "name": "value", "elementType": "$inputVec4" }, | |
| "partial_out", | |
| "partial_stats", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, | |
| { | |
| "name": "windowSize", | |
| "type": "u32", | |
| "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" | |
| } | |
| ] | |
| } | |
| ], | |
| "dispatch": { "x": "qkvNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "merge", | |
| "name": "GroupQueryAttention.DecodeSplitKMerge", | |
| "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"" }, | |
| "bindings": ["partial_out_merge", "partial_stats_merge", "output_merge"], | |
| "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_flash_cluster", | |
| "priority": 22, | |
| "when": ["flashShapeOk", "qkvSgClusterOk or qkvNoSgClusterStorageOk", "qkvPrefillClusterWorkgroupSize <= device.limits.maxComputeInvocationsPerWorkgroup", "qkvPrefillClusterWorkgroupSize <= device.limits.maxComputeWorkgroupSizeX", "headDimPlan % (4 * qkvPrefillLanesPerQuery) == 0", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "not causalNoWindow"], | |
| "demoteWhen": ["device.features.has(\"subgroups\") and not qkvF32ClusterRegisterGeometry"], | |
| "derive": { | |
| "useSubgroups": "qkvSgClusterOk", | |
| "usesF16": "qkvUsesF16", | |
| "scalar": "qkvScalar", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "qkvInputVec4", | |
| "outputElement": "qkvInputVec4", | |
| "presentElement": "qkvCacheVec4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "kvHiddenV4": "dim(shapes.keyT, 2) / 4", | |
| "presentVec4": true, | |
| "TILE_Q": "qkvPrefillQueryTile", | |
| "TILE_K": "qkvClusterTileK if useSubgroups else qkvNoSgTileK", | |
| "LPQ": "qkvPrefillLanesPerQuery", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Flash", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, | |
| "bindings": ["query_query_t", "key_key_t", "value_value_t", "output_output_t", "params_qkv_present"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_bias_flash_prefill", | |
| "priority": 35, | |
| "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedMaskSgPrefillOk or cachedNoSgClusterStorageOk", "standardSoftmax", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "not present.headSinkT"], | |
| "derive": { | |
| "useSubgroups": "cachedMaskSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasMask": true, | |
| "stageMask": "useSubgroups", | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "biasPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "attn_mask", "output_merge", "params_attn_flash_prefill_cluster"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_qnorm_rotary_flash_prefill", | |
| "priority": 37, | |
| "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "qPrepScratchFits", "standardSoftmax", "present.cosCacheT and rotaryRequested", "present.qNormWeightT", "not present.attentionBiasT", "not present.headSinkT"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "hasRotary": true, | |
| "half": "gqaHeadDim / 2", | |
| "qHeads": "attrs.num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "hasQNorm": true, | |
| "qkEps": "attrs.qk_norm_epsilon", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", | |
| "qNormScalar": "\"f16\" if tensorDtypes.qNormWeightT == \"float16\" else \"f32\"", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "qPrep", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0), dim(shapes.queryT, 1), dim(shapes.queryT, 2)]" | |
| } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "qprep", | |
| "name": "GroupQueryAttention.QPrep", | |
| "shader": "gqa-qprep.wgsl.jinja", | |
| "bindings": ["query_qprep", "cos_cache", "sin_cache", "q_norm_weight", "qout", "params_qprep"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": [ | |
| { "scratch": "qPrep", "name": "query", "buffer": "read-only-storage", "elementType": "vec4<f32>" }, | |
| "key_attention", | |
| "value_attention", | |
| "output_merge", | |
| "params_past_kv_flash_prefill" | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_decode_splitk", | |
| "priority": 38, | |
| "when": ["sharedKvFloatOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "subgroupSplitK", | |
| "cacheSeqlens": true, | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "numSplits": "cachedNumSplits", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "partialOut", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" | |
| }, | |
| { | |
| "id": "partialStats", | |
| "dtype": "float32", | |
| "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" | |
| } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "split_attention", | |
| "name": "GroupQueryAttention.DecodeSplitK", | |
| "shader": "attn-flash-decode-splitk.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "partial_out", "partial_stats", "seqlens_k", "params_split_attention"], | |
| "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "merge", | |
| "name": "GroupQueryAttention.DecodeSplitKMerge", | |
| "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["partial_out_merge", "partial_stats_merge", "output_merge"], | |
| "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_past_decode_splitk", | |
| "priority": 38, | |
| "when": ["newKvPastOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "subgroupSplitK", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "numSplits": "cachedNumSplits", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "partialOut", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" | |
| }, | |
| { | |
| "id": "partialStats", | |
| "dtype": "float32", | |
| "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" | |
| } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Merge", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "params_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "split_attention", | |
| "name": "GroupQueryAttention.DecodeSplitK", | |
| "shader": "attn-flash-decode-splitk.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "partial_out", "partial_stats", "params_split_attention"], | |
| "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "merge", | |
| "name": "GroupQueryAttention.DecodeSplitKMerge", | |
| "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["partial_out_merge", "partial_stats_merge", "output_merge"], | |
| "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_decode_splitk", | |
| "priority": 38, | |
| "when": ["windowShiftOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "subgroupSplitK", | |
| "cacheSeqlens": true, | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "numSplits": "cachedNumSplits", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "partialOut", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" | |
| }, | |
| { | |
| "id": "partialStats", | |
| "dtype": "float32", | |
| "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" | |
| } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "split_attention", | |
| "name": "GroupQueryAttention.DecodeSplitK", | |
| "shader": "attn-flash-decode-splitk.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "partial_out", "partial_stats", "seqlens_k", "params_split_attention"], | |
| "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "merge", | |
| "name": "GroupQueryAttention.DecodeSplitKMerge", | |
| "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["partial_out_merge", "partial_stats_merge", "output_merge"], | |
| "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_flash_prefill", | |
| "priority": 34, | |
| "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasMask": false, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "output_merge", "params_past_kv_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_past_flash_prefill", | |
| "priority": 34, | |
| "when": ["newKvPastOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasMask": false, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Merge", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "params_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "output_merge", "params_new_kv_past_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_rotary_flash_prefill", | |
| "priority": 36, | |
| "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "standardSoftmax", "present.cosCacheT and rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "not present.headSinkT"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasMask": false, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "hasRotary": true, | |
| "fusedRotary": true, | |
| "half": "gqaHeadDim / 2", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "cos_cache", "sin_cache", "output_merge", "params_past_kv_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_softcap_flash_prefill", | |
| "priority": 34, | |
| "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "noAuxAttentionInputs", "attrs.softcap > 0", "attrs.smooth_softmax != 1"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasSoftcap": true, | |
| "hasMask": false, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "output_merge", "params_past_kv_softcap_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_headsink_flash_prefill", | |
| "priority": 34, | |
| "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "present.headSinkT", "standardSoftmax"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasHeadSink": true, | |
| "hasMask": false, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "head_sink", "output_merge", "params_past_kv_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_bias_headsink_flash_prefill", | |
| "priority": 34, | |
| "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedMaskSgPrefillOk or cachedNoSgClusterStorageOk", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "present.headSinkT", "standardSoftmax"], | |
| "derive": { | |
| "useSubgroups": "cachedMaskSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasHeadSink": true, | |
| "hasMask": true, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "attn_mask", "head_sink", "output_merge", "params_attn_flash_prefill_cluster"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_flash_prefill", | |
| "priority": 34, | |
| "when": ["windowShiftOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "hasMask": false, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "useSeqlens": true, | |
| "zeroScalar": "gqaScalar", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "output_merge", "seqlens_k", "params_new_kv_past_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_flash_q32_broadcast", | |
| "priority": 30, | |
| "when": ["wave32SubgroupsUsable", "device.features.has(\"shader-f16\")", "tensorDtypes.queryT == \"float16\"", "flashShapeOk", "headDimPlan % 32 == 0", "headDimPlan >= 64", "headDimPlan <= 256", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "attrs.local_window_size <= 0", "not causalNoWindow or dim(shapes.queryT, 1) == dim(shapes.keyT, 1)"], | |
| "demoteWhen": ["causalNoWindow and device.features.has(\"subgroups\")"], | |
| "requires": { "features": ["subgroups"] }, | |
| "derive": { | |
| "hasBias": false, | |
| "hasCausal": "causalNoWindow", | |
| "fixedCausal": "causalNoWindow", | |
| "usesF16": true, | |
| "scalar": "\"f16\"", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "\"vec4<f16>\"", | |
| "outputElement": "\"vec4<f16>\"", | |
| "presentElement": "qkvCacheVec4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "kvHiddenV4": "dim(shapes.keyT, 2) / 4", | |
| "presentVec4": true, | |
| "kStep": 64, | |
| "qkGroups": 16 | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashQ32Broadcast", | |
| "shader": "attn-flash-q32-broadcast.wgsl.jinja", | |
| "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, | |
| "bindings": [ | |
| "query_query_t", | |
| "key_key_t", | |
| "value_value_t", | |
| "output_output_t", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), 32)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qkv_present_flash_q32_shared", | |
| "priority": 29, | |
| "when": ["tensorDtypes.queryT == \"float16\"", "flashShapeOk", "headDimPlan % 4 == 0", "headDimPlan >= 64", "headDimPlan <= 256", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "attrs.local_window_size <= 0", "ceilDiv(dim(shapes.queryT, 1), 64) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "(headDimPlan / 4) * 32 * 16 <= device.limits.maxComputeWorkgroupStorageSize", "not causalNoWindow or dim(shapes.queryT, 1) == dim(shapes.keyT, 1)"], | |
| "demoteWhen": ["causalNoWindow and device.features.has(\"subgroups\")"], | |
| "requires": { "features": ["shader-f16"] }, | |
| "derive": { | |
| "hasBias": false, | |
| "hasCausal": "causalNoWindow", | |
| "fixedCausal": "causalNoWindow", | |
| "usesF16": true, | |
| "scalar": "\"f16\"", | |
| "presentScalar": "qkvCacheScalar", | |
| "inputElement": "\"vec4<f16>\"", | |
| "outputElement": "\"vec4<f16>\"", | |
| "presentElement": "qkvCacheVec4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", | |
| "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", | |
| "kvHidden": "dim(shapes.keyT, 2)", | |
| "kvHiddenV4": "dim(shapes.keyT, 2) / 4", | |
| "presentVec4": true, | |
| "kStep": 32, | |
| "qkGroups": 8, | |
| "qStep": 64 | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashQ32Shared", | |
| "shader": "attn-flash-q32-broadcast.wgsl.jinja", | |
| "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"", "useSubgroups": "false" }, | |
| "bindings": [ | |
| "query_query_t", | |
| "key_key_t", | |
| "value_value_t", | |
| "output_output_t", | |
| { | |
| "name": "params", | |
| "struct": [ | |
| { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, | |
| { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, | |
| { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } | |
| ] | |
| } | |
| ], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), 64)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["key_key_t", "value_value_t", "present_key", "present_value", "params"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv", | |
| "priority": 30, | |
| "when": ["sharedKvFloatOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "gqaHeadDim % 2 == 0"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"copy\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache_past_kv", "vcache_past_kv", "output_past_kv", "seqlens_k", "params_past_kv"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_rotary", | |
| "priority": 31, | |
| "when": ["sharedKvFloatOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "present.cosCacheT and rotaryRequested", "not present.qNormWeightT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "half": "gqaHeadDim / 2", | |
| "hasRotary": true, | |
| "mode": "\"copy\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar", | |
| "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache_past_kv", "vcache_past_kv", "cos_cache", "sin_cache", "output_past_kv", "seqlens_k", "params_past_kv"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_qnorm_rotary", | |
| "priority": 32, | |
| "when": ["sharedKvFloatOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "present.cosCacheT and rotaryRequested", "present.qNormWeightT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "half": "gqaHeadDim / 2", | |
| "qkEps": "attrs.qk_norm_epsilon", | |
| "hasRotary": true, | |
| "hasQNorm": true, | |
| "mode": "\"copy\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar", | |
| "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", | |
| "qNormScalar": "\"f16\" if tensorDtypes.qNormWeightT == \"float16\" else \"f32\"" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache_past_kv", "vcache_past_kv", "cos_cache", "sin_cache", "q_norm_weight", "output_past_kv", "seqlens_k", "params_past_kv"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_rotary_append", | |
| "priority": 31, | |
| "when": ["windowShiftRotaryOk", "gqaScalarDispatchOk", "not present.qNormWeightT", "not present.attentionBiasT", "not present.headSinkT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "inputScalar": "\"f32\"", | |
| "zeroScalar": "\"f32\"", | |
| "windowOrigin": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_shift", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shiftRegion": "\"shift\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.WindowShiftAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shiftRegion": "\"append\"", "hasRotary": true, "half": "gqaHeadDim / 2" }, | |
| "bindings": ["new_k", "new_v", "cos_cache_f32", "sin_cache_f32", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "derive": { "hasRotary": true, "half": "gqaHeadDim / 2" }, | |
| "bindings": ["query_qprep", "kcache", "vcache", "cos_cache_f32", "sin_cache_f32", "output_past_kv", "seqlens_k", "params_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_rotary_headsink_append", | |
| "priority": 31, | |
| "when": ["windowShiftRotaryOk", "gqaScalarDispatchOk", "not present.qNormWeightT", "not present.attentionBiasT", "present.headSinkT", "tensorDtypes.headSinkT == tensorDtypes.queryT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "inputScalar": "\"f32\"", | |
| "zeroScalar": "\"f32\"", | |
| "windowOrigin": true, | |
| "hasHeadSink": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_shift", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shiftRegion": "\"shift\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.WindowShiftAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shiftRegion": "\"append\"", "hasRotary": true, "half": "gqaHeadDim / 2" }, | |
| "bindings": ["new_k", "new_v", "cos_cache_f32", "sin_cache_f32", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "derive": { "hasRotary": true, "half": "gqaHeadDim / 2" }, | |
| "bindings": ["query_qprep", "kcache", "vcache", "cos_cache_f32", "sin_cache_f32", "head_sink", "output_past_kv", "seqlens_k", "params_gqa_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_rotary_bias_append", | |
| "priority": 31, | |
| "when": ["windowShiftRotaryOk", "gqaScalarDispatchOk", "not present.qNormWeightT", "not present.headSinkT", "present.attentionBiasT", "tensorDtypes.attentionBiasT == tensorDtypes.queryT", "dim(shapes.attentionBiasT, 2) == dim(shapes.queryT, 1)", "dim(shapes.attentionBiasT, 0) == dim(shapes.queryT, 0) or dim(shapes.attentionBiasT, 0) == 1", "dim(shapes.attentionBiasT, 1) == attrs.num_heads or dim(shapes.attentionBiasT, 1) == 1"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "inputScalar": "\"f32\"", | |
| "zeroScalar": "\"f32\"", | |
| "windowOrigin": true, | |
| "hasBias": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_shift", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shiftRegion": "\"shift\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.WindowShiftAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shiftRegion": "\"append\"", "hasRotary": true, "half": "gqaHeadDim / 2" }, | |
| "bindings": ["new_k", "new_v", "cos_cache_f32", "sin_cache_f32", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "derive": { "hasRotary": true, "half": "gqaHeadDim / 2" }, | |
| "bindings": ["query_qprep", "kcache", "vcache", "cos_cache_f32", "sin_cache_f32", "attn_bias", "output_past_kv", "seqlens_k", "params_window_bias"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_past", | |
| "priority": 30, | |
| "when": ["newKvPastOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "gqaHeadDim % 2 == 0"], | |
| "derive": { | |
| "headDim": "gqaHeadDim", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge\"", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Merge", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "params_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache", "vcache", "output_past_kv", "params_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_append", | |
| "priority": 30, | |
| "when": ["windowShiftOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "gqaHeadDim % 2 == 0"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache", "vcache", "output_past_kv", "seqlens_k", "params_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_headsink_append", | |
| "priority": 30, | |
| "when": ["windowShiftOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "not present.qNormWeightT", "not present.attentionBiasT", "present.headSinkT", "tensorDtypes.headSinkT == tensorDtypes.queryT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar", | |
| "hasHeadSink": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache", "vcache", "head_sink", "output_past_kv", "seqlens_k", "params_gqa_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "window_shift_bias_append", | |
| "priority": 30, | |
| "when": ["windowShiftOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "not present.qNormWeightT", "not present.headSinkT", "present.attentionBiasT", "tensorDtypes.attentionBiasT == tensorDtypes.queryT", "dim(shapes.attentionBiasT, 2) == dim(shapes.queryT, 1)", "dim(shapes.attentionBiasT, 0) == dim(shapes.queryT, 0) or dim(shapes.attentionBiasT, 0) == 1", "dim(shapes.attentionBiasT, 1) == attrs.num_heads or dim(shapes.attentionBiasT, 1) == 1"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"window_shift\"", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar", | |
| "hasBias": true, | |
| "windowOrigin": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.WindowShift", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query_qprep", "kcache", "vcache", "attn_bias", "output_past_kv", "seqlens_k", "params_window_bias"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "new_kv_qnorm_rotary", | |
| "priority": 33, | |
| "when": ["copyWorkgroupOk", "gqaScalarDispatchOk", "present.seqlensKT", "not present.pastKeyT", "dim(shapes.keyT, 1) > 0", "tensorDtypes.queryT == \"float32\"", "not present.kScaleT", "present.cosCacheT and rotaryRequested", "present.qNormWeightT", "dim(shapes.queryT, 2) % attrs.num_heads == 0", "gqaHeadDim % 2 == 0"], | |
| "derive": { | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "half": "gqaHeadDim / 2", | |
| "qkEps": "attrs.qk_norm_epsilon", | |
| "hasRotary": true, | |
| "hasQNorm": true, | |
| "mode": "\"build\"", | |
| "hasKNorm": true | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k_f32", "src_v_f32", "cos_cache_f32", "sin_cache_f32", "k_norm_weight", "present_key_f32", "present_value_f32", "params_count_seq_past_seq"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query", "kcache_f32", "vcache_f32", "cos_cache_f32", "sin_cache_f32", "q_norm_weight_f32", "output", "params_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_bias", | |
| "priority": 30, | |
| "when": ["sharedKvCacheOk", "gqaScalarDispatchOk", "tensorDtypes.queryT == \"float32\"", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "not present.headSinkT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "hasBias": true, | |
| "mode": "\"copy\"", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query", "kcache_f32_read", "vcache_f32_read", "attn_bias", "output", "seqlens_k", "params_past_kv_bias"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_headsink", | |
| "priority": 30, | |
| "when": ["sharedKvCacheOk", "gqaScalarDispatchOk", "tensorDtypes.queryT == \"float32\"", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "present.headSinkT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "hasHeadSink": true, | |
| "mode": "\"copy\"", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query", "kcache_f32_read", "vcache_f32_read", "head_sink", "output", "seqlens_k", "params_gqa_attention"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "past_kv_bias_headsink", | |
| "priority": 30, | |
| "when": ["sharedKvCacheOk", "gqaScalarDispatchOk", "tensorDtypes.queryT == \"float32\"", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "present.headSinkT"], | |
| "derive": { | |
| "useSeqlens": true, | |
| "headDim": "gqaHeadDim", | |
| "cooperative": "gqaCoop", | |
| "dispatchUnits": "gqaDispatchUnits", | |
| "qHeads": "attrs.num_heads", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "packed": "gqaHeadDim", | |
| "hasBias": true, | |
| "hasHeadSink": true, | |
| "mode": "\"copy\"", | |
| "inputScalar": "gqaScalar" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.Attn", | |
| "shader": "gqa-attention.wgsl.jinja", | |
| "bindings": ["query", "kcache_f32_read", "vcache_f32_read", "attn_bias", "head_sink", "output", "seqlens_k", "params_past_kv_bias_headsink"], | |
| "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } | |
| }, | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.Present", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k", "src_v", "present_key_present", "present_value_present", "params_count"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "quant_int8_flash_prefill", | |
| "priority": 37, | |
| "when": ["quantizedPromptOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "attrs.kv_cache_bit_width == 8", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": false, | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "quantCacheFormat": "\"int8\"", | |
| "quantCacheElement": "\"i32\"", | |
| "quantCacheReadElement": "\"vec4<i32>\"", | |
| "packed": "gqaHeadDim", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "mode": "\"build_quant\"", | |
| "bits": 8, | |
| "qmax": 127, | |
| "qmin": -128, | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.QuantPresent", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k_f32", "src_v_f32", "k_scale", "v_scale", "present_key_quant_int8", "present_value_quant_int8", "params_quant_int8"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.QuantFlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_present_key_t", "value_present_value_t", "k_scale", "v_scale", "output_merge", "params_quant_int8_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "quant_int4_flash_prefill", | |
| "priority": 37, | |
| "when": ["quantizedPromptOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "attrs.kv_cache_bit_width == 4", "present.kScaleT", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": false, | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "quantCacheFormat": "\"int4\"", | |
| "quantCacheElement": "\"u32\"", | |
| "quantCacheReadElement": "\"u32\"", | |
| "packed": "gqaHeadDim / 2", | |
| "qHidden": "dim(shapes.queryT, 2)", | |
| "mode": "\"build_quant\"", | |
| "bits": 4, | |
| "qmax": 7, | |
| "qmin": -8, | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present", | |
| "name": "GroupQueryAttention.QuantPresent", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "bindings": ["src_k_f32", "src_v_f32", "k_scale", "v_scale", "present_key_quant_int8", "present_value_quant_int8", "params_quant_int8"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.QuantFlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_present_key_t", "value_present_value_t", "k_scale", "v_scale", "output_merge", "params_quant_int8_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "share_append_split_decode_splitk", | |
| "priority": 39, | |
| "when": ["shareAppendOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "subgroupSplitK", | |
| "cacheSeqlens": "true", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "numSplits": "cachedNumSplits", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge_share\"", | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "zeroScalar": "gqaScalar" | |
| }, | |
| "intermediates": [ | |
| { | |
| "id": "partialOut", | |
| "dtype": "float32", | |
| "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" | |
| }, | |
| { | |
| "id": "partialStats", | |
| "dtype": "float32", | |
| "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" | |
| } | |
| ], | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "split_attention", | |
| "name": "GroupQueryAttention.DecodeSplitK", | |
| "shader": "attn-flash-decode-splitk.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "partial_out", "partial_stats", "seqlens_k", "params_split_attention"], | |
| "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| }, | |
| { | |
| "id": "merge", | |
| "name": "GroupQueryAttention.DecodeSplitKMerge", | |
| "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", | |
| "derive": { "layout": "\"bhsd\"" }, | |
| "bindings": ["partial_out_merge", "partial_stats_merge", "output_merge"], | |
| "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "share_append_split_flash_prefill", | |
| "priority": 35, | |
| "when": ["shareAppendOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": true, | |
| "causalRightAlign": true, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge_share\"", | |
| "useSeqlens": true, | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "output_merge", "seqlens_k", "params_new_kv_past_flash_prefill"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "share_append_bidirectional_flash_prefill", | |
| "priority": 36, | |
| "when": ["shareAppendBidirectionalOk", "prefillFlashShapeOk", "cachedSgPrefillOk or cachedNoSgClusterStorageOk", "plainAttentionOptions"], | |
| "derive": { | |
| "useSubgroups": "cachedSgPrefillOk", | |
| "scalar": "gqaScalar", | |
| "usesF16": "tensorDtypes.queryT == \"float16\"", | |
| "hasCausal": false, | |
| "causalRightAlign": false, | |
| "headDim": "gqaHeadDim", | |
| "headDimV4": "gqaHeadDim / 4", | |
| "qHiddenV4": "dim(shapes.queryT, 2) / 4", | |
| "kvNumHeads": "attrs.kv_num_heads", | |
| "TILE_Q": "cachedPrefillQueryTile", | |
| "TILE_K": "cachedClusterTileK if useSubgroups else cachedNoSgTileK", | |
| "LPQ": "cachedPrefillLanesPerQuery", | |
| "kvHeads": "attrs.kv_num_heads", | |
| "packed": "gqaHeadDim", | |
| "mode": "\"merge_share\"", | |
| "useSeqlens": true, | |
| "inputVec4": "gqaInputVec4", | |
| "inputScalar": "gqaScalar", | |
| "batchNoSgReduction": "not useSubgroups", | |
| "shareNoSgProb": "batchNoSgReduction and not usesF16 and LPQ >= 4 and TILE_K >= 2 * LPQ" | |
| }, | |
| "passes": [ | |
| { | |
| "id": "present_retain", | |
| "name": "GroupQueryAttention.MergeShareRetain", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"retain\"" }, | |
| "bindings": ["past_k", "past_v", "present_key_present", "present_value_present", "seqlens_k", "params__uniform"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| }, | |
| "viewAlias": [ | |
| { "input": "past_k", "output": "present_key" }, | |
| { "input": "past_v", "output": "present_value" } | |
| ] | |
| }, | |
| { | |
| "id": "present_append", | |
| "name": "GroupQueryAttention.MergeShareAppend", | |
| "shader": "gqa-present.wgsl.jinja", | |
| "derive": { "shareRegion": "\"append\"" }, | |
| "bindings": ["new_k", "new_v", "present_key_present", "present_value_present", "seqlens_k", "params_gqa_present"], | |
| "dispatch": { | |
| "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", | |
| "z": 1 | |
| } | |
| }, | |
| { | |
| "id": "attention", | |
| "name": "GroupQueryAttention.FlashPrefill", | |
| "shader": "attn-flash-prefill-cluster.wgsl.jinja", | |
| "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, | |
| "bindings": ["query_attention", "key_attention", "value_attention", "output_merge", "seqlens_k", "params_new_kv_past_flash_bidirectional"], | |
| "dispatch": { | |
| "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", | |
| "y": "attrs.num_heads", | |
| "z": "dim(shapes.queryT, 0)" | |
| } | |
| } | |
| ] | |
| } | |
| ] | |
| } | |