{ "domain": "com.microsoft", "name": "GroupQueryAttention", "sinceVersion": 1, "inputs": { "queryT": { "onnx": "query", "dtype": "T", "rank": 3 }, "keyT": { "onnx": "key", "dtype": "T", "rank": 3 }, "valueT": { "onnx": "value", "dtype": "T", "rank": 3 }, "pastKeyT": { "onnx": "past_key", "dtype": "T_CACHE", "rank": 4, "optional": true }, "pastValueT": { "onnx": "past_value", "dtype": "T_CACHE", "rank": 4, "optional": true }, "seqlensKT": { "onnx": "seqlens_k", "dtype": "M", "rank": 1 }, "totalSequenceLengthT": { "onnx": "total_sequence_length", "dtype": "M", "rank": 1 }, "cosCacheT": { "onnx": "cos_cache", "dtype": "T", "rank": 2, "optional": true }, "sinCacheT": { "onnx": "sin_cache", "dtype": "T", "rank": 2, "optional": true }, "attentionBiasT": { "onnx": "attention_bias", "dtype": "T", "rank": 4, "optional": true }, "headSinkT": { "onnx": "head_sink", "dtype": "T", "rank": 1, "optional": true }, "kScaleT": { "onnx": "k_scale", "dtype": "T_KV_SCALE", "rank": 1, "optional": true }, "vScaleT": { "onnx": "v_scale", "dtype": "T_KV_SCALE", "rank": 1, "optional": true }, "qNormWeightT": { "onnx": "q_norm_weight", "dtype": "T", "rank": 1, "optional": true }, "kNormWeightT": { "onnx": "k_norm_weight", "dtype": "T", "rank": 1, "optional": true } }, "outputs": { "outputT": { "onnx": "output", "dtype": "T", "rank": 3, "shape": "shapes.queryT" }, "presentKeyT": { "onnx": "present_key", "dtype": "T_CACHE", "rank": 4 }, "presentValueT": { "onnx": "present_value", "dtype": "T_CACHE", "rank": 4 } }, "attributes": { "causal": { "default": 1 }, "do_rotary": { "default": 0 }, "k_quant_type": { "default": "NONE" }, "local_window_size": { "default": -1 }, "qk_norm_epsilon": { "default": 0.000001 }, "sliding_window_cache": { "default": 0 }, "smooth_softmax": { "default": -1 }, "softcap": { "default": 0 }, "v_quant_type": { "default": "NONE" }, "kv_cache_bit_width": {}, "kv_num_heads": {}, "num_heads": {}, "scale": {} }, "attributeConstraints": { "causal": { "values": [0, 1] }, "do_rotary": { "values": [0, 1] }, "k_quant_type": { "values": ["NONE", "PER_TENSOR", "PER_CHANNEL"] }, "kv_cache_bit_width": { "values": [4, 8] }, "kv_num_heads": { "required": true }, "num_heads": { "required": true }, "sliding_window_cache": { "values": [0, 1] }, "v_quant_type": { "values": ["NONE", "PER_TENSOR", "PER_CHANNEL"] } }, "typeConstraints": { "T": ["float32", "float16"], "T_CACHE": ["float32", "float16", "uint8", "int8"], "T_KV_SCALE": ["float32"], "M": ["int32"] }, "tunables": { "WORKGROUP_SIZE": { "default": 256 }, "COPY_WORKGROUP_SIZE": { "default": 64 }, "SCALAR_WORKGROUP_SIZE": { "default": 64 }, "COOPERATIVE_QUERY_THRESHOLD": { "default": 6144 }, "MAX_SPLITS": { "default": 16 }, "QKV_SPLIT_TILE_K": { "default": 128 }, "CACHED_SPLIT_TILE_K": { "default": 256 }, "FLASH_MIN_HEAD_DIM": { "default": 32 }, "FLASH_MAX_HEAD_DIM": { "default": 256 }, "PREFILL_QUERY_TILE": { "default": 32 }, "PREFILL_LANES_PER_QUERY": { "default": 8 }, "PREFILL_QUERY_TILE_F16": { "default": 16 }, "PREFILL_LANES_PER_QUERY_F16": { "default": 8 }, "QKV_PREFILL_MIN_QUERY_TOKENS": { "default": 31 }, "QKV_DECODE_MIN_KV_TOKENS": { "default": 512 }, "CACHED_FLASH_MIN_HEAD_DIM": { "default": 16 }, "CACHED_DECODE_MIN_KV_TOKENS": { "default": 1024 }, "CLUSTER_TILE_K_F32": { "default": 8 }, "CLUSTER_TILE_K_F16": { "default": 8 }, "NO_SG_TILE_K_MAX": { "default": 16 }, "COOPERATIVE_WORKGROUP_SIZE": { "default": 32 }, "CLUSTER_MAX_SLICE": { "default": 8 }, "MATERIALIZED_SGMAT_QUERY_TILE": { "default": 64 }, "MATERIALIZED_SGMAT_KEY_TILE": { "default": 64 }, "MATERIALIZED_SGMAT_INNER_TILE": { "default": 32 }, "MATERIALIZED_SOFTMAX_WORKGROUP_SIZE": { "default": 256 }, "MATERIALIZED_SGMAT_MIN_SEQ": { "default": 256 }, "MATERIALIZED_SGMAT_PAST_MIN_SEQ": { "default": 256 }, "MATERIALIZED_SGMAT_WINDOW_MIN_QSEQ": { "default": 128 }, "MATERIALIZED_SGMAT_APPEND_MIN_QSEQ": { "default": 128 } }, "derive": { "deviceWorkgroupCap": "min(device.limits.maxComputeInvocationsPerWorkgroup, device.limits.maxComputeWorkgroupSizeX)", "wave32Adapter": "has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize == 32 and device.adapterInfo.subgroupMaxSize == 32", "subgroupsWave32": "device.features.has(\"subgroups\") and wave32Adapter", "narrowSubgroupRange": "has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize < device.adapterInfo.subgroupMaxSize and device.adapterInfo.subgroupMaxSize <= 16", "canPinSubgroupSize32": "device.features.has(\"subgroups\") and device.features.has(\"subgroup-size-control\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize <= 32 and device.adapterInfo.subgroupMaxSize >= 32", "pinSubgroupSize32": "canPinSubgroupSize32 and not wave32Adapter", "wave32Effective": "wave32Adapter or pinSubgroupSize32", "windowCacheRequested": "attrs.sliding_window_cache == 1", "headDimPlan": "dim(shapes.queryT, 2) / attrs.num_heads if (ranks.queryT == 3 and attrs.num_heads > 0) else 0", "copyWorkgroupSize": "min(tunables.COPY_WORKGROUP_SIZE, min(device.limits.maxComputeInvocationsPerWorkgroup, device.limits.maxComputeWorkgroupSizeX))", "scalarWorkgroupSize": "min(tunables.SCALAR_WORKGROUP_SIZE, min(device.limits.maxComputeInvocationsPerWorkgroup, device.limits.maxComputeWorkgroupSizeX))", "copyWorkgroupOk": "copyWorkgroupSize >= 1", "scalarWorkgroupOk": "scalarWorkgroupSize >= 1", "metadataInputsOk": "present.seqlensKT and present.totalSequenceLengthT", "unquantizedCacheAttrs": "attrs.k_quant_type == \"NONE\" and attrs.v_quant_type == \"NONE\" and not attrs.kv_cache_bit_width and not present.kScaleT and not present.vScaleT", "qkNormInputsOk": "present.qNormWeightT == present.kNormWeightT", "rotaryRequested": "attrs.do_rotary == 1", "noAuxAttentionInputs": "not present.cosCacheT and not rotaryRequested and not present.qNormWeightT and not present.attentionBiasT and not present.headSinkT", "standardSoftmax": "attrs.softcap == 0 and attrs.smooth_softmax != 1", "plainAttentionOptions": "noAuxAttentionInputs and standardSoftmax", "qkvMaskModeOk": "(attrs.causal == 0 and attrs.local_window_size == -1) or (attrs.causal == 1 and attrs.local_window_size > 0)", "qkvContractOk": "plainAttentionOptions and not windowCacheRequested and copyWorkgroupOk and metadataInputsOk and unquantizedCacheAttrs and qkNormInputsOk and qkvMaskModeOk and (not present.pastKeyT) and (not present.pastValueT) and ranks.queryT == 3 and ranks.keyT == 3 and ranks.valueT == 3 and ranks.outputT == 3 and ranks.presentKeyT == 4 and ranks.presentValueT == 4 and (tensorDtypes.queryT == \"float32\" or tensorDtypes.queryT == \"float16\") and f16Ok(tensorDtypes.queryT) and tensorDtypes.keyT == tensorDtypes.queryT and tensorDtypes.valueT == tensorDtypes.queryT and tensorDtypes.outputT == tensorDtypes.queryT and (tensorDtypes.presentKeyT == \"float32\" or tensorDtypes.presentKeyT == \"float16\") and f16Ok(tensorDtypes.presentKeyT) and tensorDtypes.presentValueT == tensorDtypes.presentKeyT and attrs.num_heads > 0 and attrs.kv_num_heads > 0 and attrs.num_heads % attrs.kv_num_heads == 0 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.keyT, 2) == attrs.kv_num_heads * headDimPlan and dim(shapes.valueT, 2) == dim(shapes.keyT, 2) and dim(shapes.queryT, 0) == dim(shapes.keyT, 0) and dim(shapes.queryT, 0) == dim(shapes.valueT, 0) and dim(shapes.keyT, 1) == dim(shapes.valueT, 1) and dim(shapes.outputT, 0) == dim(shapes.queryT, 0) and dim(shapes.outputT, 1) == dim(shapes.queryT, 1) and dim(shapes.outputT, 2) == dim(shapes.queryT, 2) and dim(shapes.presentKeyT, 0) == dim(shapes.queryT, 0) and dim(shapes.presentKeyT, 1) == attrs.kv_num_heads and dim(shapes.presentKeyT, 2) == dim(shapes.keyT, 1) and dim(shapes.presentKeyT, 3) == headDimPlan and dim(shapes.presentValueT, 0) == dim(shapes.presentKeyT, 0) and dim(shapes.presentValueT, 1) == dim(shapes.presentKeyT, 1) and dim(shapes.presentValueT, 2) == dim(shapes.presentKeyT, 2) and dim(shapes.presentValueT, 3) == dim(shapes.presentKeyT, 3)", "flashWorkgroupOk": "tunables.WORKGROUP_SIZE <= deviceWorkgroupCap", "flashShapeOk": "qkvContractOk and flashWorkgroupOk and headDimPlan % 4 == 0 and headDimPlan >= tunables.FLASH_MIN_HEAD_DIM and headDimPlan <= tunables.FLASH_MAX_HEAD_DIM and (dim(shapes.queryT, 1) * attrs.num_heads >= 248 or (dim(shapes.queryT, 1) == 1 and dim(shapes.keyT, 1) >= tunables.QKV_DECODE_MIN_KV_TOKENS) or (dim(shapes.queryT, 1) > 1 and dim(shapes.keyT, 1) >= 2048)) and dim(shapes.queryT, 0) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and dim(shapes.queryT, 1) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "decodeSplitKShapeOk": "qkvContractOk and flashWorkgroupOk and dim(shapes.queryT, 1) == 1 and dim(shapes.keyT, 1) >= tunables.QKV_DECODE_MIN_KV_TOKENS and headDimPlan % 4 == 0 and headDimPlan >= tunables.FLASH_MIN_HEAD_DIM and headDimPlan <= tunables.FLASH_MAX_HEAD_DIM and dim(shapes.queryT, 0) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "fallbackShapeOk": "qkvContractOk and headDimPlan <= deviceWorkgroupCap and headDimPlan <= tunables.FLASH_MAX_HEAD_DIM", "gqaHeadDim": "dim(shapes.queryT, 2) / attrs.num_heads", "qkvScalar": "\"f16\" if tensorDtypes.queryT == \"float16\" else \"f32\"", "qkvInputVec4": "\"vec4\" if tensorDtypes.queryT == \"float16\" else \"vec4\"", "qkvCacheScalar": "\"f16\" if tensorDtypes.presentKeyT == \"float16\" else \"f32\"", "qkvCacheVec4": "\"vec4\" if tensorDtypes.presentKeyT == \"float16\" else \"vec4\"", "qkvUsesF16": "tensorDtypes.queryT == \"float16\" or tensorDtypes.presentKeyT == \"float16\"", "gqaScalar": "\"f16\" if tensorDtypes.queryT == \"float16\" else \"f32\"", "gqaInputVec4": "\"vec4\" if tensorDtypes.queryT == \"float16\" else \"vec4\"", "qkvNumSplits": "min(tunables.MAX_SPLITS, ceilDiv(dim(shapes.keyT, 1), tunables.QKV_SPLIT_TILE_K))", "cachedNumSplits": "min(tunables.MAX_SPLITS, ceilDiv(dim(shapes.presentKeyT, 2), tunables.CACHED_SPLIT_TILE_K))", "cachedBaseOk": "attrs.causal == 1 and metadataInputsOk and qkNormInputsOk", "sharedKvCacheOk": "cachedBaseOk and unquantizedCacheAttrs and not windowCacheRequested and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) == 0 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim", "gqaFlashHd": "gqaHeadDim >= tunables.CACHED_FLASH_MIN_HEAD_DIM and gqaHeadDim <= tunables.FLASH_MAX_HEAD_DIM", "f16OrF32": "(tensorDtypes.queryT == \"float32\" or tensorDtypes.queryT == \"float16\") and f16Ok(tensorDtypes.queryT) and tensorDtypes.keyT == tensorDtypes.queryT and tensorDtypes.valueT == tensorDtypes.queryT and tensorDtypes.outputT == tensorDtypes.queryT and tensorDtypes.presentKeyT == tensorDtypes.queryT and tensorDtypes.presentValueT == tensorDtypes.queryT and ((not present.pastKeyT) or tensorDtypes.pastKeyT == tensorDtypes.queryT) and ((not present.pastValueT) or tensorDtypes.pastValueT == tensorDtypes.queryT)", "subgroupCluster8": "device.features.has(\"subgroups\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize >= 8 and device.adapterInfo.subgroupMinSize % 8 == 0 and device.adapterInfo.subgroupMaxSize % 8 == 0", "qkvPrefillQueryTile": "tunables.PREFILL_QUERY_TILE_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_QUERY_TILE", "qkvPrefillLanesPerQuery": "max(tunables.PREFILL_LANES_PER_QUERY_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_LANES_PER_QUERY, ceilDiv(headDimPlan / 4, tunables.CLUSTER_MAX_SLICE))", "cachedPrefillQueryTile": "tunables.PREFILL_QUERY_TILE_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_QUERY_TILE", "cachedPrefillLanesPerQuery": "max(tunables.PREFILL_LANES_PER_QUERY_F16 if tensorDtypes.queryT == \"float16\" else tunables.PREFILL_LANES_PER_QUERY, ceilDiv(gqaHeadDim / 4, tunables.CLUSTER_MAX_SLICE))", "subgroupQkvPrefillCluster": "not narrowSubgroupRange and device.features.has(\"subgroups\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize >= qkvPrefillLanesPerQuery and device.adapterInfo.subgroupMinSize % qkvPrefillLanesPerQuery == 0 and device.adapterInfo.subgroupMaxSize % qkvPrefillLanesPerQuery == 0", "subgroupCachedPrefillCluster": "not narrowSubgroupRange and device.features.has(\"subgroups\") and has(device.adapterInfo, \"subgroupMinSize\") and has(device.adapterInfo, \"subgroupMaxSize\") and device.adapterInfo.subgroupMinSize >= cachedPrefillLanesPerQuery and device.adapterInfo.subgroupMinSize % cachedPrefillLanesPerQuery == 0 and device.adapterInfo.subgroupMaxSize % cachedPrefillLanesPerQuery == 0", "subgroupSplitK": "subgroupCluster8 and device.adapterInfo.subgroupMaxSize <= tunables.WORKGROUP_SIZE and tunables.WORKGROUP_SIZE % device.adapterInfo.subgroupMinSize == 0 and tunables.WORKGROUP_SIZE % device.adapterInfo.subgroupMaxSize == 0", "gqaQueryCount": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)", "cooperativeWorkgroupOk": "tunables.COOPERATIVE_WORKGROUP_SIZE <= deviceWorkgroupCap", "gqaCoopWorkgroupBytes": "((tunables.COOPERATIVE_WORKGROUP_SIZE + 1) * gqaHeadDim + tunables.COOPERATIVE_WORKGROUP_SIZE * 2) * 4", "gqaCoop": "cooperativeWorkgroupOk and gqaQueryCount < tunables.COOPERATIVE_QUERY_THRESHOLD and gqaCoopWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", "gqaScalarDispatchOk": "gqaCoop or scalarWorkgroupOk", "gqaDispatchUnits": "gqaQueryCount if gqaCoop else ceilDiv(gqaQueryCount, scalarWorkgroupSize)", "gqaDispatchFits": "dim(shapes.queryT, 0) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "sharedKvFloatOk": "sharedKvCacheOk and f16OrF32", "newKvPastOk": "cachedBaseOk and unquantizedCacheAttrs and not windowCacheRequested and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) > 0 and f16OrF32 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim and dim(shapes.presentKeyT, 2) == dim(shapes.pastKeyT, 2) + dim(shapes.keyT, 1)", "windowCapacity": "dim(shapes.pastKeyT, 2) if present.pastKeyT else 0", "windowFitsCapacity": "attrs.local_window_size <= windowCapacity if attrs.local_window_size > 0 else true", "windowShiftOk": "cachedBaseOk and unquantizedCacheAttrs and not present.cosCacheT and not rotaryRequested and windowCacheRequested and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) > 0 and f16OrF32 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim and dim(shapes.presentKeyT, 2) == windowCapacity and dim(shapes.keyT, 1) <= windowCapacity and windowFitsCapacity", "shareAppendOk": "cachedBaseOk and unquantizedCacheAttrs and not windowCacheRequested and copyWorkgroupOk and present.pastKeyT and present.pastValueT and dim(shapes.keyT, 1) > 0 and f16OrF32 and dim(shapes.queryT, 2) % attrs.num_heads == 0 and dim(shapes.pastKeyT, 1) == attrs.kv_num_heads and dim(shapes.pastKeyT, 3) == gqaHeadDim and dim(shapes.presentKeyT, 2) == dim(shapes.pastKeyT, 2) and dim(shapes.keyT, 1) <= dim(shapes.pastKeyT, 2)", "decodeFlashShapeOk": "flashWorkgroupOk and dim(shapes.queryT, 1) == 1 and gqaHeadDim % 4 == 0 and gqaFlashHd and dim(shapes.presentKeyT, 2) >= tunables.CACHED_DECODE_MIN_KV_TOKENS and gqaDispatchFits", "qkvPrefillClusterWorkgroupSize": "qkvPrefillQueryTile * qkvPrefillLanesPerQuery", "cachedPrefillClusterWorkgroupSize": "cachedPrefillQueryTile * cachedPrefillLanesPerQuery", "cachedPrefillClusterWorkgroupOk": "cachedPrefillClusterWorkgroupSize <= deviceWorkgroupCap", "prefillFlashShapeOk": "cachedPrefillClusterWorkgroupOk and gqaHeadDim % (4 * cachedPrefillLanesPerQuery) == 0 and gqaFlashHd and dim(shapes.queryT, 1) >= cachedPrefillQueryTile and gqaDispatchFits and ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "quantScaleShapeOk": "((attrs.k_quant_type == \"PER_TENSOR\" and numel(shapes.kScaleT) == 1) or (attrs.k_quant_type == \"PER_CHANNEL\" and numel(shapes.kScaleT) == attrs.kv_num_heads * gqaHeadDim)) and ((attrs.v_quant_type == \"PER_TENSOR\" and numel(shapes.vScaleT) == 1) or (attrs.v_quant_type == \"PER_CHANNEL\" and numel(shapes.vScaleT) == attrs.kv_num_heads * gqaHeadDim))", "quantizedNewKvOk": "cachedBaseOk and not windowCacheRequested and copyWorkgroupOk and present.kScaleT and present.vScaleT and attrs.k_quant_type == attrs.v_quant_type and attrs.k_quant_type != \"NONE\" and (attrs.kv_cache_bit_width == 4 or attrs.kv_cache_bit_width == 8) and quantScaleShapeOk and tensorDtypes.queryT == \"float32\" and tensorDtypes.keyT == \"float32\" and tensorDtypes.valueT == \"float32\" and tensorDtypes.outputT == \"float32\" and tensorDtypes.presentKeyT == (\"uint8\" if attrs.kv_cache_bit_width == 4 else \"int8\") and tensorDtypes.presentValueT == tensorDtypes.presentKeyT and dim(shapes.presentKeyT, 3) == gqaHeadDim / (8 / attrs.kv_cache_bit_width) and dim(shapes.presentValueT, 3) == dim(shapes.presentKeyT, 3) and dim(shapes.keyT, 1) > 0 and gqaHeadDim % 2 == 0", "quantizedPromptOk": "quantizedNewKvOk and dim(shapes.keyT, 1) == dim(shapes.presentKeyT, 2)", "quantizedDecodeOptionsOk": "plainAttentionOptions and not present.sinCacheT and not present.kNormWeightT", "quantizedCachedDecodeOk": "not windowCacheRequested and quantizedNewKvOk and quantizedDecodeOptionsOk and attrs.kv_cache_bit_width == 8 and present.pastKeyT and present.pastValueT and dim(shapes.queryT, 1) == 1 and dim(shapes.keyT, 1) == 1 and tensorDtypes.pastKeyT == \"int8\" and tensorDtypes.pastValueT == \"int8\" and tensorDtypes.presentKeyT == \"int8\" and tensorDtypes.presentValueT == \"int8\" and dim(shapes.pastKeyT, 0) == dim(shapes.presentKeyT, 0) and dim(shapes.pastKeyT, 1) == dim(shapes.presentKeyT, 1) and dim(shapes.pastKeyT, 2) == dim(shapes.presentKeyT, 2) and dim(shapes.pastKeyT, 3) == dim(shapes.presentKeyT, 3)", "qkvSplitScratchBytes": "dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits * headDimPlan * 4", "qkvSplitStatsBytes": "2 * dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits * 4", "qkvSplitScratchFits": "qkvSplitScratchBytes <= device.limits.maxStorageBufferBindingSize and qkvSplitScratchBytes <= device.limits.maxBufferSize and qkvSplitStatsBytes <= device.limits.maxStorageBufferBindingSize and qkvSplitStatsBytes <= device.limits.maxBufferSize", "cachedSplitScratchBytes": "dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits * gqaHeadDim * 4", "cachedSplitStatsBytes": "2 * dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits * 4", "cachedSplitScratchFits": "cachedSplitScratchBytes <= device.limits.maxStorageBufferBindingSize and cachedSplitScratchBytes <= device.limits.maxBufferSize and cachedSplitStatsBytes <= device.limits.maxStorageBufferBindingSize and cachedSplitStatsBytes <= device.limits.maxBufferSize", "qPrepScratchBytes": "numel(shapes.queryT) * 4", "qPrepScratchFits": "qPrepScratchBytes <= device.limits.maxStorageBufferBindingSize and qPrepScratchBytes <= device.limits.maxBufferSize", "qkvTiledWorkgroupBytes": "dim(shapes.valueT, 2) / attrs.kv_num_heads * 32 * 4", "qkvTiledStorageOk": "qkvTiledWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", "qkvClusterTileK": "tunables.CLUSTER_TILE_K_F32 if tensorDtypes.queryT == \"float32\" and headDimPlan <= tunables.QKV_SPLIT_TILE_K else tunables.CLUSTER_TILE_K_F16", "qkvClusterWorkgroupBytes": "qkvClusterTileK * headDimPlan * (8 if tensorDtypes.queryT == \"float32\" else 4)", "qkvClusterStorageOk": "qkvClusterWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", "qkvF32ClusterRegisterGeometry": "tensorDtypes.queryT != \"float32\" or (headDimPlan % 4 == 0 and headDimPlan / (4 * qkvPrefillLanesPerQuery) <= tunables.CLUSTER_MAX_SLICE)", "cachedF32ClusterRegisterGeometry": "tensorDtypes.queryT != \"float32\" or (gqaHeadDim % 4 == 0 and gqaHeadDim / (4 * cachedPrefillLanesPerQuery) <= tunables.CLUSTER_MAX_SLICE)", "qkvNoSgReductionBytesPerKey": "qkvPrefillClusterWorkgroupSize * 4", "cachedNoSgReductionBytesPerKey": "cachedPrefillClusterWorkgroupSize * 4", "qkvNoSgTileBytes": "headDimPlan * (8 if tensorDtypes.queryT == \"float32\" else 4) + qkvNoSgReductionBytesPerKey", "qkvNoSgTileK": "min(tunables.NO_SG_TILE_K_MAX, max(1, floor(device.limits.maxComputeWorkgroupStorageSize / qkvNoSgTileBytes)))", "qkvNoSgWorkgroupBytes": "qkvNoSgTileK * qkvNoSgTileBytes", "qkvNoSgClusterStorageOk": "qkvNoSgWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", "cachedClusterTileK": "tunables.CLUSTER_TILE_K_F32 if tensorDtypes.queryT == \"float32\" and gqaHeadDim <= tunables.QKV_SPLIT_TILE_K else tunables.CLUSTER_TILE_K_F16", "cachedClusterWorkgroupBytes": "cachedClusterTileK * gqaHeadDim * (8 if tensorDtypes.queryT == \"float32\" else 4)", "cachedClusterStorageOk": "cachedClusterWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", "cachedMaskClusterWorkgroupBytes": "cachedClusterWorkgroupBytes + cachedPrefillQueryTile * cachedClusterTileK * 4", "cachedMaskClusterStorageOk": "cachedMaskClusterWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", "cachedNoSgTileBytes": "gqaHeadDim * (8 if tensorDtypes.queryT == \"float32\" else 4) + cachedNoSgReductionBytesPerKey", "cachedNoSgTileK": "min(tunables.NO_SG_TILE_K_MAX, max(1, floor(device.limits.maxComputeWorkgroupStorageSize / cachedNoSgTileBytes)))", "cachedNoSgWorkgroupBytes": "cachedNoSgTileK * cachedNoSgTileBytes", "cachedNoSgClusterStorageOk": "cachedNoSgWorkgroupBytes <= device.limits.maxComputeWorkgroupStorageSize", "gqaMatQueryTile": "tunables.MATERIALIZED_SGMAT_QUERY_TILE", "gqaMatKeyTile": "tunables.MATERIALIZED_SGMAT_KEY_TILE", "gqaMatInnerTile": "tunables.MATERIALIZED_SGMAT_INNER_TILE", "gqaMatSubgroupRows": "floor(gqaMatQueryTile / 16)", "gqaMatSubgroupCols": "floor(gqaMatKeyTile / 32)", "gqaMatWorkgroupSize": "gqaMatSubgroupRows * gqaMatSubgroupCols * 32", "gqaMatCompactStorageBytes": "(gqaMatQueryTile + gqaMatKeyTile) * gqaMatInnerTile * 4", "gqaMatGeometryOk": "gqaMatQueryTile >= 16 and gqaMatQueryTile % 16 == 0 and gqaMatKeyTile >= 32 and gqaMatKeyTile <= 64 and gqaMatKeyTile % 32 == 0 and gqaMatInnerTile == 32 and gqaMatQueryTile >= 64 and gqaMatKeyTile >= 64", "gqaMatResourcesFit": "gqaMatGeometryOk and gqaMatWorkgroupSize <= deviceWorkgroupCap and gqaMatCompactStorageBytes <= device.limits.maxComputeWorkgroupStorageSize", "gqaMatScoreBytes": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.keyT, 1) * 4", "gqaMatScoreFits": "gqaMatScoreBytes <= device.limits.maxStorageBufferBindingSize and gqaMatScoreBytes <= device.limits.maxBufferSize", "gqaMatBuffersFit": "numel(shapes.queryT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.keyT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.valueT) * 4 <= device.limits.maxStorageBufferBindingSize", "gqaMatDispatchFits": "ceilDiv(dim(shapes.keyT, 1), gqaMatKeyTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and dim(shapes.queryT, 0) * attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "gqaMatStatSlots": "ceilDiv(dim(shapes.keyT, 1), gqaMatKeyTile) * gqaMatSubgroupCols", "gqaMatRowStatsWg": "min(tunables.MATERIALIZED_SOFTMAX_WORKGROUP_SIZE, deviceWorkgroupCap)", "gqaMatRowStatsElements": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * 2", "gqaMatScorePartialElements": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatStatSlots * dim(shapes.queryT, 1) * 2", "gqaMatCoreOk": "qkvContractOk and attrs.causal == 0 and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and headDimPlan >= 64 and headDimPlan <= 256 and headDimPlan % 16 == 0 and dim(shapes.queryT, 1) >= tunables.MATERIALIZED_SGMAT_MIN_SEQ and dim(shapes.keyT, 1) >= tunables.MATERIALIZED_SGMAT_MIN_SEQ and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatScoreFits and gqaMatBuffersFit and gqaMatResourcesFit and gqaMatDispatchFits", "gqaMatPastQSeq": "dim(shapes.queryT, 1)", "gqaMatPastKvSeq": "dim(shapes.presentKeyT, 2)", "gqaMatPastScoreBytes": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatPastQSeq * gqaMatPastKvSeq * 4", "gqaMatPastScoreFits": "gqaMatPastScoreBytes <= device.limits.maxStorageBufferBindingSize and gqaMatPastScoreBytes <= device.limits.maxBufferSize", "gqaMatPastBuffersFit": "numel(shapes.queryT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.presentKeyT) * 4 <= device.limits.maxStorageBufferBindingSize and numel(shapes.presentValueT) * 4 <= device.limits.maxStorageBufferBindingSize", "gqaMatPastStatSlots": "ceilDiv(gqaMatPastKvSeq, gqaMatKeyTile) * gqaMatSubgroupCols", "gqaMatPastRowStatsElements": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatPastQSeq * 2", "gqaMatPastScorePartialElements": "dim(shapes.queryT, 0) * attrs.num_heads * gqaMatPastStatSlots * gqaMatPastQSeq * 2", "gqaMatPastDispatchFits": "ceilDiv(gqaMatPastKvSeq, gqaMatKeyTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and ceilDiv(gqaMatPastQSeq, gqaMatQueryTile) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535) and dim(shapes.queryT, 0) * attrs.num_heads <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "gqaMatShareAppendOk": "shareAppendOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and dim(shapes.queryT, 1) == dim(shapes.keyT, 1) and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and dim(shapes.queryT, 1) >= tunables.MATERIALIZED_SGMAT_APPEND_MIN_QSEQ and dim(shapes.presentKeyT, 2) >= dim(shapes.queryT, 1) and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", "gqaMatWindowOk": "windowShiftOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size > 0 and dim(shapes.queryT, 1) == dim(shapes.keyT, 1) and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and dim(shapes.queryT, 1) >= tunables.MATERIALIZED_SGMAT_WINDOW_MIN_QSEQ and windowCapacity >= dim(shapes.queryT, 1) and attrs.local_window_size + dim(shapes.queryT, 1) >= windowCapacity and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", "gqaMatPastCoreOk": "sharedKvFloatOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", "gqaMatPastRotaryOk": "sharedKvFloatOk and standardSoftmax and present.cosCacheT and rotaryRequested and not present.qNormWeightT and not present.attentionBiasT and not present.headSinkT and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaHeadDim % 8 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", "gqaMatNewPastOk": "newKvPastOk and plainAttentionOptions and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", "gqaMatPastCoreF16Ok": "sharedKvFloatOk and plainAttentionOptions and tensorDtypes.queryT == \"float16\" and device.features.has(\"shader-f16\") and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", "gqaMatPastRotaryF16Ok": "sharedKvFloatOk and standardSoftmax and present.cosCacheT and rotaryRequested and not present.qNormWeightT and not present.attentionBiasT and not present.headSinkT and tensorDtypes.queryT == \"float16\" and device.features.has(\"shader-f16\") and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaHeadDim % 8 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits", "gqaMatPastBiasOk": "sharedKvFloatOk and standardSoftmax and not present.cosCacheT and not rotaryRequested and not present.qNormWeightT and not present.headSinkT and present.attentionBiasT and ranks.attentionBiasT == 4 and tensorDtypes.attentionBiasT == \"float32\" and (dim(shapes.attentionBiasT, 0) == dim(shapes.queryT, 0) or dim(shapes.attentionBiasT, 0) == 1) and (dim(shapes.attentionBiasT, 1) == attrs.num_heads or dim(shapes.attentionBiasT, 1) == 1) and dim(shapes.attentionBiasT, 2) == dim(shapes.queryT, 1) and dim(shapes.attentionBiasT, 3) == dim(shapes.presentKeyT, 2) and tensorDtypes.queryT == \"float32\" and attrs.local_window_size == -1 and gqaHeadDim >= 64 and gqaHeadDim <= 256 and gqaHeadDim % 16 == 0 and gqaMatPastQSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= tunables.MATERIALIZED_SGMAT_PAST_MIN_SEQ and gqaMatPastKvSeq >= gqaMatPastQSeq and device.features.has(\"subgroups\") and wave32Effective and device.features.has(\"chromium-experimental-subgroup-matrix\") and gqaMatPastScoreFits and gqaMatPastBuffersFit and gqaMatResourcesFit and gqaMatPastDispatchFits" }, "bindings": { "query": { "arg": "queryT", "buffer": "read-only-storage", "elementType": "f32" }, "key": { "arg": "keyT", "buffer": "read-only-storage", "elementType": "f32" }, "scores": { "scratch": "materializedScores", "buffer": "storage", "elementType": "f32" }, "scorePartials": { "scratch": "materializedScorePartials", "buffer": "storage", "elementType": "f32" }, "scorePartials_2": { "scratch": "materializedScorePartials", "name": "scorePartials", "buffer": "read-only-storage", "elementType": "f32" }, "rowStats": { "scratch": "materializedRowStats", "buffer": "storage", "elementType": "f32" }, "scores_2": { "scratch": "materializedScores", "name": "scores", "buffer": "read-only-storage", "elementType": "f32" }, "value": { "arg": "valueT", "buffer": "read-only-storage", "elementType": "f32" }, "rowStats_2": { "scratch": "materializedRowStats", "name": "rowStats", "buffer": "read-only-storage", "elementType": "f32" }, "output": { "arg": "outputT", "buffer": "storage", "elementType": "f32" }, "key_2": { "arg": "keyT", "name": "key", "buffer": "read-only-storage", "elementType": "$inputElement" }, "value_2": { "arg": "valueT", "name": "value", "buffer": "read-only-storage", "elementType": "$inputElement" }, "present_key": { "arg": "presentKeyT", "buffer": "storage", "elementType": "$presentElement" }, "present_value": { "arg": "presentValueT", "buffer": "storage", "elementType": "$presentElement" }, "params": { "buffer": "uniform", "struct": [ { "name": "batchSize", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } ] }, "src_k": { "arg": "pastKeyT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "src_v": { "arg": "pastValueT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "present_key_2": { "arg": "presentKeyT", "name": "present_key", "buffer": "storage", "elementType": "$inputScalar" }, "present_value_2": { "arg": "presentValueT", "name": "present_value", "buffer": "storage", "elementType": "$inputScalar" }, "params_5": { "name": "params", "buffer": "uniform", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)" } ] }, "key_3": { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "f32" }, "params_6": { "name": "params", "buffer": "uniform", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } ] }, "params_7": { "name": "params", "buffer": "uniform", "struct": [ { "name": "rows", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)" } ] }, "value_3": { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "f32" }, "params_8": { "name": "params", "buffer": "uniform", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" } ] }, "key_4": { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "f16" }, "value_4": { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "f16" }, "output_2": { "arg": "outputT", "name": "output", "buffer": "storage", "elementType": "f16" }, "query_3": { "arg": "queryT", "name": "query", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "cos_cache": { "arg": "cosCacheT", "buffer": "read-only-storage", "elementType": "$cosScalar" }, "sin_cache": { "arg": "sinCacheT", "buffer": "read-only-storage", "elementType": "$cosScalar" }, "qout": { "scratch": "qPrep", "buffer": "storage", "elementType": "vec4" }, "params_10": { "name": "params", "buffer": "uniform", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" } ] }, "past_k": { "arg": "pastKeyT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "past_v": { "arg": "pastValueT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "new_k": { "arg": "keyT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "new_v": { "arg": "valueT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "params_11": { "name": "params", "buffer": "uniform", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)" }, { "name": "seq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "pastSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "keySeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } ] }, "seqlens_k": { "arg": "seqlensKT", "buffer": "read-only-storage", "elementType": "i32" }, "params_12": { "name": "params", "buffer": "uniform", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)" }, { "name": "seq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "keySeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } ] }, "params_15": { "name": "params", "buffer": "uniform", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, { "name": "seq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "keySeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } ] }, "kcache": { "arg": "presentKeyT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "vcache": { "arg": "presentValueT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "output_3": { "arg": "outputT", "name": "output", "buffer": "storage", "elementType": "$inputScalar" }, "params_16": { "name": "params", "buffer": "uniform", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" } ] }, "head_sink": { "arg": "headSinkT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "cos_cache_2": { "arg": "cosCacheT", "name": "cos_cache", "buffer": "read-only-storage", "elementType": "f32" }, "sin_cache_2": { "arg": "sinCacheT", "name": "sin_cache", "buffer": "read-only-storage", "elementType": "f32" }, "query_6": { "arg": "queryT", "name": "query", "buffer": "read-only-storage", "elementType": "$inputElement" }, "output_4": { "arg": "outputT", "name": "output", "buffer": "storage", "elementType": "$outputElement" }, "params_19": { "name": "params", "buffer": "uniform", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" } ] }, "src_k_2": { "arg": "keyT", "name": "src_k", "buffer": "read-only-storage", "elementType": "f32" }, "src_v_2": { "arg": "valueT", "name": "src_v", "buffer": "read-only-storage", "elementType": "f32" }, "k_scale": { "arg": "kScaleT", "buffer": "read-only-storage", "elementType": "f32" }, "v_scale": { "arg": "vScaleT", "buffer": "read-only-storage", "elementType": "f32" }, "present_key_3": { "arg": "presentKeyT", "name": "present_key", "buffer": "storage", "elementType": "$quantCacheElement" }, "present_value_3": { "arg": "presentValueT", "name": "present_value", "buffer": "storage", "elementType": "$quantCacheElement" }, "params_20": { "name": "params", "buffer": "uniform", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, { "name": "seq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] }, "partial_out": { "scratch": "partialOut", "buffer": "storage", "elementType": "vec4" }, "partial_stats": { "scratch": "partialStats", "buffer": "storage", "elementType": "vec2" }, "partial_out_2": { "scratch": "partialOut", "name": "partial_out", "buffer": "read-only-storage", "elementType": "vec4" }, "partial_stats_2": { "scratch": "partialStats", "name": "partial_stats", "buffer": "read-only-storage", "elementType": "vec2" }, "output_5": { "arg": "outputT", "name": "output", "buffer": "storage", "elementType": "$inputVec4" }, "query_8": { "arg": "queryT", "name": "query", "buffer": "read-only-storage", "elementType": "$inputVec4" }, "key_7": { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "$inputVec4" }, "value_7": { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "$inputVec4" }, "attn_mask": { "arg": "attentionBiasT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "params_25": { "name": "params", "buffer": "uniform", "struct": [ { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "isCausal", "type": "u32", "value": 1 }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "maskBatchStride", "type": "u32", "value": "0 if dim(shapes.attentionBiasT, 0) == 1 else dim(shapes.attentionBiasT, 1) * dim(shapes.attentionBiasT, 2) * dim(shapes.attentionBiasT, 3)" }, { "name": "maskHeadStride", "type": "u32", "value": "0 if dim(shapes.attentionBiasT, 1) == 1 else dim(shapes.attentionBiasT, 2) * dim(shapes.attentionBiasT, 3)" }, { "name": "maskSeqStride", "type": "u32", "value": "dim(shapes.attentionBiasT, 3)" } ] }, "q_norm_weight": { "arg": "qNormWeightT", "buffer": "read-only-storage", "elementType": "$qNormScalar" }, "params_26": { "name": "params", "buffer": "uniform", "struct": [ { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "isCausal", "type": "u32", "value": 1 }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" } ] }, "params_27": { "name": "params", "buffer": "uniform", "struct": [ { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" } ] }, "params_28": { "name": "params", "buffer": "uniform", "struct": [ { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "isCausal", "type": "u32", "value": 1 }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" } ] }, "params_29": { "name": "params", "buffer": "uniform", "struct": [ { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "isCausal", "type": "u32", "value": 1 }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" } ] }, "kcache_3": { "arg": "pastKeyT", "name": "kcache", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "vcache_3": { "arg": "pastValueT", "name": "vcache", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "params_31": { "name": "params", "buffer": "uniform", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" } ] }, "k_norm_weight": { "arg": "kNormWeightT", "buffer": "read-only-storage", "elementType": "f32" }, "present_key_5": { "arg": "presentKeyT", "name": "present_key", "buffer": "storage", "elementType": "f32" }, "present_value_5": { "arg": "presentValueT", "name": "present_value", "buffer": "storage", "elementType": "f32" }, "params_32": { "name": "params", "buffer": "uniform", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, { "name": "seq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "pastSeq", "type": "u32", "value": 0 } ] }, "kcache_4": { "arg": "presentKeyT", "name": "kcache", "buffer": "read-only-storage", "elementType": "f32" }, "vcache_4": { "arg": "presentValueT", "name": "vcache", "buffer": "read-only-storage", "elementType": "f32" }, "q_norm_weight_2": { "arg": "qNormWeightT", "name": "q_norm_weight", "buffer": "read-only-storage", "elementType": "f32" }, "kcache_5": { "arg": "pastKeyT", "name": "kcache", "buffer": "read-only-storage", "elementType": "f32" }, "vcache_5": { "arg": "pastValueT", "name": "vcache", "buffer": "read-only-storage", "elementType": "f32" }, "attn_bias": { "arg": "attentionBiasT", "buffer": "read-only-storage", "elementType": "$inputScalar" }, "params_33": { "name": "params", "buffer": "uniform", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" }, { "name": "biasBatch", "type": "u32", "value": "dim(shapes.attentionBiasT, 0)" }, { "name": "biasHeads", "type": "u32", "value": "dim(shapes.attentionBiasT, 1)" } ] }, "params_34": { "name": "params", "buffer": "uniform", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" } ] }, "params_35": { "name": "params", "buffer": "uniform", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.pastKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, { "name": "biasBatch", "type": "u32", "value": "dim(shapes.attentionBiasT, 0)" }, { "name": "biasHeads", "type": "u32", "value": "dim(shapes.attentionBiasT, 1)" } ] }, "key_8": { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "$quantCacheReadElement" }, "value_8": { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "$quantCacheReadElement" }, "params_36": { "name": "params", "buffer": "uniform", "struct": [ { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "isCausal", "type": "u32", "value": 1 }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] } }, "variants": [ { "id": "qkv_present_materialized_sgmat_f32", "priority": 23, "when": ["gqaMatCoreOk"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "kvHidden": "dim(shapes.keyT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "presentScalar": "qkvCacheScalar", "presentElement": "qkvCacheVec4", "presentVec4": true, "inputElement": "qkvInputVec4", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "kvHiddenV4": "dim(shapes.keyT, 2) / 4" }, "intermediates": [ { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.keyT, 1)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatScorePartialElements]" } ], "passes": [ { "id": "scores", "name": "GroupQueryAttention.MaterializedScoresSgmat", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"" }, "bindings": [ "query", "key", "scores", "scorePartials", { "name": "params", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if has(attrs, \"scale\") else 0" } ] } ], "dispatch": { "x": "ceilDiv(dim(shapes.keyT, 1), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.MaterializedRowStatsCombine", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": [ "scorePartials_2", "rowStats", { "name": "params", "struct": [ { "name": "rows", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)" } ] } ], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.MaterializedApplySgmat", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"" }, "bindings": [ "scores_2", "value", "rowStats_2", "output", { "name": "params", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" } ] } ], "dispatch": { "x": "ceilDiv(headDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } }, { "id": "present", "name": "GroupQueryAttention.MaterializedPresent", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "past_kv_materialized_sgmat_f32", "priority": 35, "when": ["gqaMatPastCoreOk"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"copy\"", "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "usesF16": false }, "intermediates": [ { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "scores", "name": "GroupQueryAttention.PastMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["query", "key_3", "scores", "scorePartials", "params_6"], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.PastMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.PastMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["scores_2", "value_3", "rowStats_2", "output", "params_8"], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "past_kv_bias_materialized_sgmat_f32", "priority": 36, "when": ["gqaMatPastBiasOk"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"copy\"", "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "usesF16": false }, "intermediates": [ { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "scores", "name": "GroupQueryAttention.PastMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true, "scoreBias": true }, "bindings": [ "query", "key_3", { "arg": "attentionBiasT", "name": "attention_bias", "elementType": "f32" }, "scores", "scorePartials", { "name": "params", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "biasBatch", "type": "u32", "value": "dim(shapes.attentionBiasT, 0)" }, { "name": "biasHeads", "type": "u32", "value": "dim(shapes.attentionBiasT, 1)" } ] } ], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.PastMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.PastMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["scores_2", "value_3", "rowStats_2", "output", "params_8"], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "past_kv_materialized_sgmat_f16", "priority": 35, "when": ["gqaMatPastCoreF16Ok"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f16", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"copy\"", "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "usesF16": true, "operandF16": true }, "intermediates": [ { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "scores", "name": "GroupQueryAttention.PastMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": [ { "arg": "queryT", "name": "query", "elementType": "f16" }, "key_4", "scores", "scorePartials", "params_6" ], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.PastMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.PastMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["scores_2", "value_4", "rowStats_2", "output_2", "params_8"], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "past_kv_rotary_materialized_sgmat_f32", "priority": 37, "when": ["gqaMatPastRotaryOk"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"copy\"", "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "usesF16": false, "half": "gqaHeadDim / 2", "qHeads": "attrs.num_heads", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", "hasQNorm": false }, "intermediates": [ { "id": "qPrep", "dtype": "float32", "shape": "[dim(shapes.queryT, 0), dim(shapes.queryT, 1), dim(shapes.queryT, 2)]" }, { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "qprep", "name": "GroupQueryAttention.RotaryQPrep", "shader": "gqa-qprep.wgsl.jinja", "bindings": ["query_3", "cos_cache", "sin_cache", "qout", "params_10"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "scores", "name": "GroupQueryAttention.PastRotaryMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": [ { "scratch": "qPrep", "name": "query", "buffer": "read-only-storage", "elementType": "f32" }, "key_3", "scores", "scorePartials", "params_6" ], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.PastRotaryMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.PastRotaryMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["scores_2", "value_3", "rowStats_2", "output", "params_8"], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "past_kv_rotary_materialized_sgmat_f16", "priority": 37, "when": ["gqaMatPastRotaryF16Ok"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f16", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"copy\"", "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "usesF16": true, "half": "gqaHeadDim / 2", "qHeads": "attrs.num_heads", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", "hasQNorm": false, "operandF16": true }, "intermediates": [ { "id": "qPrep", "dtype": "float16", "shape": "[dim(shapes.queryT, 0), dim(shapes.queryT, 1), dim(shapes.queryT, 2)]" }, { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "qprep", "name": "GroupQueryAttention.RotaryQPrep", "shader": "gqa-qprep.wgsl.jinja", "derive": { "qPrepF16": true }, "bindings": [ "query_3", "cos_cache", "sin_cache", { "scratch": "qPrep", "name": "qout", "elementType": "vec4" }, "params_10" ], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "scores", "name": "GroupQueryAttention.PastRotaryMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": [ { "scratch": "qPrep", "name": "query", "buffer": "read-only-storage", "elementType": "f16" }, "key_4", "scores", "scorePartials", "params_6" ], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.PastRotaryMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.PastRotaryMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["scores_2", "value_4", "rowStats_2", "output_2", "params_8"], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "new_kv_past_materialized_sgmat_f32", "priority": 35, "when": ["gqaMatNewPastOk"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"merge\"", "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "usesF16": false }, "intermediates": [ { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Merge", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "params_11"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "scores", "name": "GroupQueryAttention.NewPastMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["query", "key_3", "scores", "scorePartials", "params_6"], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.NewPastMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.NewPastMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true }, "bindings": ["scores_2", "value_3", "rowStats_2", "output", "params_8"], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "window_shift_materialized_sgmat_f32", "priority": 36, "when": ["gqaMatWindowOk"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"window_shift\"", "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "usesF16": false }, "intermediates": [ { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.WindowShift", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "scores", "name": "GroupQueryAttention.WindowMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true, "scoreWindow": true, "useSeqlens": true }, "bindings": [ "query", "key_3", "scores", "scorePartials", "seqlens_k", { "name": "params", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } ] } ], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.WindowMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.WindowMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true, "scoreWindow": true, "useSeqlens": true }, "bindings": [ "scores_2", "value_3", "rowStats_2", "output", "seqlens_k", { "name": "params", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size" } ] } ], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "share_append_materialized_sgmat_f32", "priority": 36, "when": ["gqaMatShareAppendOk"], "requires": { "features": ["subgroups", "chromium-experimental-subgroup-matrix"], "subgroupMatrixConfigs": [{ "componentType": "f32", "resultComponentType": "f32", "M": 8, "N": 8, "K": 8 }] }, "derive": { "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "qHidden": "dim(shapes.queryT, 2)", "hasBias": false, "useSubgroups": true, "attentionScaleExpression": "\"select(inverseSqrt(f32(HEAD_DIM)), params.scale, params.scale != 0.0)\"", "materializedSgmatQueryTile": "gqaMatQueryTile", "materializedSgmatKeyTile": "gqaMatKeyTile", "materializedSgmatInnerTile": "gqaMatInnerTile", "materializedSgmatDirectScoreStore": false, "materializedSgmatDirectApplyStore": false, "materializedSgmatRuntimeDirectStore": false, "materializedRowStatsWg": "gqaMatRowStatsWg", "statSlots": "gqaMatPastStatSlots", "statQuerySeq": "dim(shapes.queryT, 1)", "mode": "\"merge_share\"", "useSeqlens": true, "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "inputScalar": "gqaScalar", "usesF16": false }, "intermediates": [ { "id": "materializedScores", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1) * dim(shapes.presentKeyT, 2)]" }, { "id": "materializedRowStats", "dtype": "float32", "shape": "[gqaMatPastRowStatsElements]" }, { "id": "materializedScorePartials", "dtype": "float32", "shape": "[gqaMatPastScorePartialElements]" } ], "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "scores", "name": "GroupQueryAttention.ShareAppendMaterializedScores", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"score\"", "emitRowStats": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true, "useSeqlens": true }, "bindings": ["query", "key_3", "scores", "scorePartials", "seqlens_k", "params_6"], "dispatch": { "x": "ceilDiv(dim(shapes.presentKeyT, 2), gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" }, "subgroupCollectivesWidth": 32 }, { "id": "rowstats", "name": "GroupQueryAttention.ShareAppendMaterializedRowStats", "shader": "attn-materialized-rowstats-combine-f32.wgsl.jinja", "bindings": ["scorePartials_2", "rowStats", "params_7"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (gqaMatRowStatsWg)), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": 1, "z": 1 } }, { "id": "apply", "name": "GroupQueryAttention.ShareAppendMaterializedApply", "shader": "attn-materialized-sgmat-f32.wgsl.jinja", "derive": { "phase": "\"apply\"", "fusedSoftmax": true, "layout": "\"bsh\"", "kvLayout": "\"bhsd\"", "causalRightAlign": true, "useSeqlens": true }, "bindings": ["scores_2", "value_3", "rowStats_2", "output", "seqlens_k", "params_8"], "dispatch": { "x": "ceilDiv(gqaHeadDim, gqaMatKeyTile)", "y": "ceilDiv(dim(shapes.queryT, 1), gqaMatQueryTile)", "z": "dim(shapes.queryT, 0) * attrs.num_heads" } } ] }, { "id": "new_kv_share_append_split", "priority": 32, "when": ["shareAppendOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "not present.kNormWeightT", "gqaHeadDim % 2 == 0"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "mode": "\"merge_share\"", "usesF16": "tensorDtypes.queryT == \"float16\"", "inputScalar": "gqaScalar" }, "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query_3", "kcache", "vcache", "output_3", "seqlens_k", "params_16"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "new_kv_share_append_headsink_split", "priority": 30, "when": ["shareAppendOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "not present.kNormWeightT", "not present.attentionBiasT", "present.headSinkT"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "hasHeadSink": true, "mode": "\"merge_share\"", "usesF16": "tensorDtypes.queryT == \"float16\"", "inputScalar": "gqaScalar" }, "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": [ "query_3", "kcache", "vcache", "head_sink", "output_3", "seqlens_k", { "name": "params", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" } ] } ], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "new_kv_share_append_rotary_split", "priority": 30, "when": ["shareAppendOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "present.cosCacheT and rotaryRequested", "not present.qNormWeightT", "not present.kNormWeightT", "not present.attentionBiasT", "not present.headSinkT", "tensorDtypes.queryT == \"float32\""], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "half": "gqaHeadDim / 2", "hasRotary": true, "mode": "\"merge_share\"", "inputScalar": "\"f32\"", "cosScalar": "\"f32\"" }, "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "cos_cache_2", "sin_cache_2", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query_3", "kcache", "vcache", "cos_cache_2", "sin_cache_2", "output_3", "seqlens_k", "params_16"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "qkv_present_tiled_nosg", "priority": 19, "when": ["plainAttentionOptions", "flashShapeOk", "qkvTiledStorageOk", "headDimPlan % 4 == 0", "headDimPlan <= 128", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "attrs.local_window_size <= 0"], "derive": { "scalar": "qkvScalar", "inputScalar": "qkvScalar", "outputScalar": "qkvScalar", "presentScalar": "qkvCacheScalar", "inputElement": "qkvScalar", "outputElement": "qkvScalar", "presentElement": "qkvCacheScalar", "usesF16": "qkvUsesF16", "blockM": 32, "vHeadCap": "dim(shapes.valueT, 2) / attrs.kv_num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "kvHidden": "dim(shapes.keyT, 2)" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.PrefillTiledNoSg", "shader": "attention-rank4-tiled.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": [ { "arg": "queryT", "name": "q", "elementType": "$scalar" }, { "arg": "keyT", "name": "k", "elementType": "$scalar" }, { "arg": "valueT", "name": "v", "elementType": "$scalar" }, { "arg": "outputT", "name": "y", "elementType": "$scalar" }, { "name": "params", "struct": [ { "name": "count", "type": "u32", "value": "numel(shapes.outputT)" }, { "name": "qHeads", "type": "u32", "value": "attrs.num_heads" }, { "name": "kvHeads", "type": "u32", "value": "attrs.kv_num_heads" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "headSize", "type": "u32", "value": "dim(shapes.queryT, 2) / attrs.num_heads" }, { "name": "vHeadSize", "type": "u32", "value": "dim(shapes.valueT, 2) / attrs.kv_num_heads" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "0" }, { "name": "isCausal", "type": "u32", "value": "0" }, { "name": "qHidden", "type": "u32", "value": "dim(shapes.queryT, 2)" }, { "name": "kvHidden", "type": "u32", "value": "dim(shapes.keyT, 2)" }, { "name": "vHidden", "type": "u32", "value": "dim(shapes.valueT, 2)" } ] } ], "dispatch": { "x": "min(ceilDiv((dim(shapes.outputT, 0) * attrs.num_heads * ceilDiv(dim(shapes.outputT, 1), 32) * 32), (32)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.outputT, 0) * attrs.num_heads * ceilDiv(dim(shapes.outputT, 1), 32) * 32), (32)), 65535)", "z": 1 } }, { "id": "present", "name": "GroupQueryAttention.PresentTiledNoSg", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "qkv_present_flash", "priority": 20, "when": ["flashShapeOk"], "requires": { "features": ["subgroups"] }, "derive": { "headsFromParams": false, "hasBias": false, "hasCausal": false, "combineSubgroups": true, "usesF16": "qkvUsesF16", "scalar": "qkvScalar", "inputVec4": "qkvInputVec4", "outputScalar": "qkvScalar", "presentScalar": "qkvCacheScalar", "inputElement": "qkvInputVec4", "outputElement": "qkvInputVec4", "presentElement": "qkvCacheVec4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "kvHidden": "dim(shapes.keyT, 2)", "kvHiddenV4": "dim(shapes.keyT, 2) / 4", "presentVec4": true, "presentElem": "qkvCacheVec4", "hasWindow": "attrs.local_window_size > 0" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Flash", "shader": "attn-flash-online.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": ["query_6", "key_2", "value_2", "output_4", "params_19"], "dispatch": { "x": "dim(shapes.queryT, 1)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "qkv_present", "priority": 0, "when": ["fallbackShapeOk"], "derive": { "headsFromParams": false, "hasBias": false, "hasCausal": false, "hasKeyLimit": false, "scaleFallbackRsqrt": true, "usesF16": "qkvUsesF16", "inputScalar": "qkvScalar", "outputScalar": "qkvScalar", "presentScalar": "qkvCacheScalar", "inputElement": "qkvScalar", "outputElement": "qkvScalar", "presentElement": "qkvCacheScalar", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "kvHidden": "dim(shapes.keyT, 2)", "workgroupSize": "min(256, pow2ceil(dim(shapes.queryT, 2) / attrs.num_heads))", "hasWindow": "attrs.local_window_size > 0" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Attention", "shader": "attn-online-scalar.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": ["query_6", "key_2", "value_2", "output_4", "params_19"], "dispatch": { "x": "min(dim(shapes.queryT, 1), min(device.limits.maxComputeWorkgroupsPerDimension, 65535))", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT)), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "quant_int8", "priority": 35, "when": ["quantizedPromptOk", "gqaScalarDispatchOk", "attrs.kv_cache_bit_width == 8"], "derive": { "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim / (8 / 8)", "quantized": true, "bits": 8, "mode": "\"build_quant\"", "qmax": "127 if 8 == 8 else 7", "qmin": "0 - (128 if 8 == 8 else 8)", "quantCacheElement": "\"i32\" if 8 == 8 else \"u32\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k_2", "src_v_2", "k_scale", "v_scale", "present_key_3", "present_value_3", "params_20"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": [ "query", { "arg": "presentKeyT", "name": "kcache", "buffer": "read-only-storage", "elementType": "$quantCacheElement" }, { "arg": "presentValueT", "name": "vcache", "buffer": "read-only-storage", "elementType": "$quantCacheElement" }, "k_scale", "v_scale", "output", { "name": "params", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] } ], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "quant_int4", "priority": 35, "when": ["quantizedPromptOk", "gqaScalarDispatchOk", "attrs.kv_cache_bit_width == 4"], "derive": { "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim / (8 / 4)", "quantized": true, "bits": 4, "mode": "\"build_quant\"", "qmax": "127 if 4 == 8 else 7", "qmin": "0 - (128 if 4 == 8 else 8)", "quantCacheElement": "\"i32\" if 4 == 8 else \"u32\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k_2", "src_v_2", "k_scale", "v_scale", "present_key_3", "present_value_3", "params_20"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": [ "query", { "arg": "presentKeyT", "name": "kcache", "buffer": "read-only-storage", "elementType": "$quantCacheElement" }, { "arg": "presentValueT", "name": "vcache", "buffer": "read-only-storage", "elementType": "$quantCacheElement" }, "k_scale", "v_scale", "output", { "name": "params", "struct": [ { "name": "batch", "type": "u32", "value": "dim(shapes.queryT, 0)" }, { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "softcap", "type": "f32", "value": "attrs.softcap" }, { "name": "smoothSoftmax", "type": "u32", "value": "1 if attrs.smooth_softmax == 1 else 0" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] } ], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "quant_int8_decode_splitk", "priority": 45, "when": ["quantizedCachedDecodeOk", "decodeFlashShapeOk", "cachedNumSplits >= 2", "cachedSplitScratchFits", "subgroupSplitK if \"\" == \"\" else true"], "derive": { "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "numSplits": "cachedNumSplits", "usesF16": false, "scalar": "\"f32\"", "inputScalar": "\"i32\"", "inputVec4": "\"vec4\"", "quantizedCache": true, "cacheSeqlens": true, "bits": 8, "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "qmax": 127, "qmin": -128, "hasBias": false, "hasWindow": "attrs.local_window_size > 0", "useSubgroups": "\"\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits]" } ], "passes": [ { "id": "copy_past", "name": "GroupQueryAttention.QuantInt8CopyPast", "shader": "gqa-present.wgsl.jinja", "derive": { "modeSpec": "\"copy\"" }, "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "append", "name": "GroupQueryAttention.QuantInt8Append", "shader": "gqa-present.wgsl.jinja", "derive": { "modeSpec": "\"append_quant\"" }, "bindings": [ "src_k_2", "src_v_2", "k_scale", "v_scale", "seqlens_k", { "arg": "presentKeyT", "name": "present_key", "elementType": "i32" }, { "arg": "presentValueT", "name": "present_value", "elementType": "i32" }, { "name": "params", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, { "name": "seq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] } ], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.QuantInt8DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": [ { "arg": "queryT", "name": "query", "elementType": "vec4" }, { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "vec4" }, { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "vec4" }, "k_scale", "v_scale", "seqlens_k", "partial_out", "partial_stats", { "name": "params", "struct": [ { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] } ], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "merge", "name": "GroupQueryAttention.QuantInt8DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "qkv_present_flash_splitk", "priority": 25, "when": ["decodeSplitKShapeOk", "qkvSplitScratchFits", "device.features.has(\"subgroups\") if \"\" == \"\" else true"], "derive": { "combineSubgroups": "\"\" == \"\"", "scalar": "qkvScalar", "inputVec4": "qkvInputVec4", "outputScalar": "qkvScalar", "presentScalar": "qkvCacheScalar", "inputElement": "qkvInputVec4", "outputElement": "qkvInputVec4", "presentElement": "qkvCacheVec4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "kvHidden": "dim(shapes.keyT, 2)", "kvHiddenV4": "dim(shapes.keyT, 2) / 4", "numSplits": "qkvNumSplits", "presentVec4": true, "usesF16": "qkvUsesF16", "presentElem": "qkvCacheVec4", "hasWindow": "attrs.local_window_size > 0", "useSubgroups": "\"\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits * (dim(shapes.queryT, 2) / attrs.num_heads)]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits]" } ], "passes": [ { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": [ "query_8", { "arg": "keyT", "name": "key", "elementType": "$inputVec4" }, { "arg": "valueT", "name": "value", "elementType": "$inputVec4" }, "partial_out", "partial_stats", { "name": "params", "struct": [ { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" } ] } ], "dispatch": { "x": "qkvNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "qkv_present_flash_cluster", "priority": 22, "when": ["flashShapeOk", "qkvClusterStorageOk if \"\" == \"\" else qkvNoSgClusterStorageOk", "qkvPrefillClusterWorkgroupSize <= device.limits.maxComputeInvocationsPerWorkgroup", "qkvPrefillClusterWorkgroupSize <= device.limits.maxComputeWorkgroupSizeX", "headDimPlan % (4 * qkvPrefillLanesPerQuery) == 0", "qkvF32ClusterRegisterGeometry if \"\" == \"\" else true", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "subgroupQkvPrefillCluster if \"\" == \"\" else true"], "demoteWhen": ["device.features.has(\"subgroups\") and not qkvF32ClusterRegisterGeometry"], "derive": { "hasBias": false, "hasCausal": false, "headsFromParams": false, "combineSubgroups": true, "usesF16": "qkvUsesF16", "scalar": "qkvScalar", "inputVec4": "qkvInputVec4", "outputScalar": "qkvScalar", "presentScalar": "qkvCacheScalar", "inputElement": "qkvInputVec4", "outputElement": "qkvInputVec4", "presentElement": "qkvCacheVec4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "kvHidden": "dim(shapes.keyT, 2)", "kvHiddenV4": "dim(shapes.keyT, 2) / 4", "presentVec4": true, "TILE_Q": "qkvPrefillQueryTile", "TILE_K": "qkvClusterTileK", "presentElem": "qkvCacheVec4", "LPQ": "qkvPrefillLanesPerQuery", "hasWindow": "attrs.local_window_size > 0", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Flash", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": ["query_6", "key_2", "value_2", "output_4", "params_19"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "past_kv_bias_flash_prefill", "priority": 35, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedMaskClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "standardSoftmax", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "not present.headSinkT", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": true, "maskIsBool": false, "stageMask": "\"\" == \"\"", "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "attn_mask", "output_5", "params_25"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "past_kv_qnorm_rotary_flash_prefill", "priority": 37, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "qPrepScratchFits", "standardSoftmax", "present.cosCacheT and rotaryRequested", "present.qNormWeightT", "not present.attentionBiasT", "not present.headSinkT", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "hasRotary": true, "half": "gqaHeadDim / 2", "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "hasQNorm": true, "qkEps": "attrs.qk_norm_epsilon", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", "qNormScalar": "\"f16\" if tensorDtypes.qNormWeightT == \"float16\" else \"f32\"", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "intermediates": [ { "id": "qPrep", "dtype": "float32", "shape": "[dim(shapes.queryT, 0), dim(shapes.queryT, 1), dim(shapes.queryT, 2)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "qprep", "name": "GroupQueryAttention.QPrep", "shader": "gqa-qprep.wgsl.jinja", "bindings": ["query_3", "cos_cache", "sin_cache", "q_norm_weight", "qout", "params_10"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": [ { "scratch": "qPrep", "name": "query", "buffer": "read-only-storage", "elementType": "vec4" }, "key_7", "value_7", "output_5", "params_26" ], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "quant_int8_decode_splitk_nosg", "priority": 45, "when": ["quantizedCachedDecodeOk", "decodeFlashShapeOk", "cachedNumSplits >= 2", "cachedSplitScratchFits", "subgroupSplitK if \"_nosg\" == \"\" else true"], "derive": { "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "numSplits": "cachedNumSplits", "usesF16": false, "scalar": "\"f32\"", "inputScalar": "\"i32\"", "inputVec4": "\"vec4\"", "quantizedCache": true, "cacheSeqlens": true, "bits": 8, "packed": "gqaHeadDim", "kvHeads": "attrs.kv_num_heads", "qmax": 127, "qmin": -128, "hasBias": false, "hasWindow": "attrs.local_window_size > 0", "useSubgroups": "\"_nosg\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * cachedNumSplits]" } ], "passes": [ { "id": "copy_past", "name": "GroupQueryAttention.QuantInt8CopyPast", "shader": "gqa-present.wgsl.jinja", "derive": { "modeSpec": "\"copy\"" }, "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "append", "name": "GroupQueryAttention.QuantInt8Append", "shader": "gqa-present.wgsl.jinja", "derive": { "modeSpec": "\"append_quant\"" }, "bindings": [ "src_k_2", "src_v_2", "k_scale", "v_scale", "seqlens_k", { "arg": "presentKeyT", "name": "present_key", "elementType": "i32" }, { "arg": "presentValueT", "name": "present_value", "elementType": "i32" }, { "name": "params", "struct": [ { "name": "count", "type": "u32", "value": "dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)" }, { "name": "seq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "totalSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] } ], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.QuantInt8DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": [ { "arg": "queryT", "name": "query", "elementType": "vec4" }, { "arg": "presentKeyT", "name": "key", "buffer": "read-only-storage", "elementType": "vec4" }, { "arg": "presentValueT", "name": "value", "buffer": "read-only-storage", "elementType": "vec4" }, "k_scale", "v_scale", "seqlens_k", "partial_out", "partial_stats", { "name": "params", "struct": [ { "name": "kvSeq", "type": "u32", "value": "dim(shapes.presentKeyT, 2)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" }, { "name": "perChannel", "type": "u32", "value": "1 if dim(shapes.kScaleT, 0) > 1 else 0" } ] } ], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "merge", "name": "GroupQueryAttention.QuantInt8DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "qkv_present_flash_splitk_nosg", "priority": 25, "when": ["decodeSplitKShapeOk", "qkvSplitScratchFits", "device.features.has(\"subgroups\") if \"_nosg\" == \"\" else true"], "derive": { "combineSubgroups": "\"_nosg\" == \"\"", "scalar": "qkvScalar", "inputVec4": "qkvInputVec4", "outputScalar": "qkvScalar", "presentScalar": "qkvCacheScalar", "inputElement": "qkvInputVec4", "outputElement": "qkvInputVec4", "presentElement": "qkvCacheVec4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "kvHidden": "dim(shapes.keyT, 2)", "kvHiddenV4": "dim(shapes.keyT, 2) / 4", "numSplits": "qkvNumSplits", "presentVec4": true, "usesF16": "qkvUsesF16", "presentElem": "qkvCacheVec4", "hasWindow": "attrs.local_window_size > 0", "useSubgroups": "\"_nosg\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits * (dim(shapes.queryT, 2) / attrs.num_heads)]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * qkvNumSplits]" } ], "passes": [ { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": [ "query_8", { "arg": "keyT", "name": "key", "elementType": "$inputVec4" }, { "arg": "valueT", "name": "value", "elementType": "$inputVec4" }, "partial_out", "partial_stats", { "name": "params", "struct": [ { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" }, { "name": "windowSize", "type": "u32", "value": "attrs.local_window_size if attrs.local_window_size > 0 else 0" } ] } ], "dispatch": { "x": "qkvNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "qkv_present_flash_cluster_nosg", "priority": 22, "when": ["flashShapeOk", "qkvClusterStorageOk if \"_nosg\" == \"\" else qkvNoSgClusterStorageOk", "qkvPrefillClusterWorkgroupSize <= device.limits.maxComputeInvocationsPerWorkgroup", "qkvPrefillClusterWorkgroupSize <= device.limits.maxComputeWorkgroupSizeX", "headDimPlan % (4 * qkvPrefillLanesPerQuery) == 0", "qkvF32ClusterRegisterGeometry if \"_nosg\" == \"\" else true", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "subgroupQkvPrefillCluster if \"_nosg\" == \"\" else true"], "demoteWhen": ["device.features.has(\"subgroups\") and not qkvF32ClusterRegisterGeometry"], "derive": { "hasBias": false, "hasCausal": false, "headsFromParams": false, "combineSubgroups": true, "usesF16": "qkvUsesF16", "scalar": "qkvScalar", "inputVec4": "qkvInputVec4", "outputScalar": "qkvScalar", "presentScalar": "qkvCacheScalar", "inputElement": "qkvInputVec4", "outputElement": "qkvInputVec4", "presentElement": "qkvCacheVec4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "kvHidden": "dim(shapes.keyT, 2)", "kvHiddenV4": "dim(shapes.keyT, 2) / 4", "presentVec4": true, "TILE_Q": "qkvPrefillQueryTile", "TILE_K": "qkvNoSgTileK", "presentElem": "qkvCacheVec4", "LPQ": "qkvPrefillLanesPerQuery", "hasWindow": "attrs.local_window_size > 0", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Flash", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": ["query_6", "key_2", "value_2", "output_4", "params_19"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "past_kv_bias_flash_prefill_nosg", "priority": 35, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedMaskClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "standardSoftmax", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "not present.headSinkT", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": true, "maskIsBool": false, "stageMask": "\"_nosg\" == \"\"", "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "attn_mask", "output_5", "params_25"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_qnorm_rotary_flash_prefill_nosg", "priority": 37, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "qPrepScratchFits", "standardSoftmax", "present.cosCacheT and rotaryRequested", "present.qNormWeightT", "not present.attentionBiasT", "not present.headSinkT", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "hasRotary": true, "half": "gqaHeadDim / 2", "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "hasQNorm": true, "qkEps": "attrs.qk_norm_epsilon", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", "qNormScalar": "\"f16\" if tensorDtypes.qNormWeightT == \"float16\" else \"f32\"", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "intermediates": [ { "id": "qPrep", "dtype": "float32", "shape": "[dim(shapes.queryT, 0), dim(shapes.queryT, 1), dim(shapes.queryT, 2)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "qprep", "name": "GroupQueryAttention.QPrep", "shader": "gqa-qprep.wgsl.jinja", "bindings": ["query_3", "cos_cache", "sin_cache", "q_norm_weight", "qout", "params_10"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.num_heads * dim(shapes.queryT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": [ { "scratch": "qPrep", "name": "query", "buffer": "read-only-storage", "elementType": "vec4" }, "key_7", "value_7", "output_5", "params_26" ], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_decode_splitk", "priority": 38, "when": ["sharedKvFloatOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"\" == \"\" else true"], "derive": { "cacheSeqlens": true, "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"\" == \"\"", "useSubgroups": "\"\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "seqlens_k", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "new_kv_past_decode_splitk", "priority": 38, "when": ["newKvPastOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"\" == \"\"", "useSubgroups": "\"\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Merge", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "params_11"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "window_shift_decode_splitk", "priority": 38, "when": ["windowShiftOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"\" == \"\" else true"], "derive": { "cacheSeqlens": true, "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"window_shift\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"\" == \"\"", "useSubgroups": "\"\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.WindowShift", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "seqlens_k", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_decode_splitk_nosg", "priority": 38, "when": ["sharedKvFloatOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"_nosg\" == \"\" else true"], "derive": { "cacheSeqlens": true, "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"_nosg\" == \"\"", "useSubgroups": "\"_nosg\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "seqlens_k", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "new_kv_past_decode_splitk_nosg", "priority": 38, "when": ["newKvPastOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"_nosg\" == \"\"", "useSubgroups": "\"_nosg\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.Merge", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "params_11"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "window_shift_decode_splitk_nosg", "priority": 38, "when": ["windowShiftOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"_nosg\" == \"\" else true"], "derive": { "cacheSeqlens": true, "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"window_shift\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"_nosg\" == \"\"", "useSubgroups": "\"_nosg\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present", "name": "GroupQueryAttention.WindowShift", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "seqlens_k", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_flash_prefill", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "params_26"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "new_kv_past_flash_prefill", "priority": 34, "when": ["newKvPastOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Merge", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "params_11"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "params_28"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "past_kv_rotary_flash_prefill", "priority": 36, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "standardSoftmax", "present.cosCacheT and rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "not present.headSinkT", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "hasRotary": true, "fusedRotary": true, "half": "gqaHeadDim / 2", "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "cos_cache", "sin_cache", "output_5", "params_26"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "past_kv_softcap_flash_prefill", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "noAuxAttentionInputs", "attrs.softcap > 0", "attrs.smooth_softmax != 1", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasSoftcap": true, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "params_29"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "past_kv_headsink_flash_prefill", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "present.headSinkT", "standardSoftmax", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasHeadSink": true, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "head_sink", "output_5", "params_26"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "past_kv_bias_headsink_flash_prefill", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedMaskClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "present.headSinkT", "standardSoftmax", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasHeadSink": true, "hasMask": true, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "attn_mask", "head_sink", "output_5", "params_25"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "window_shift_flash_prefill", "priority": 34, "when": ["windowShiftOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"window_shift\"", "useSeqlens": true, "zeroScalar": "gqaScalar", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.WindowShift", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "seqlens_k", "params_28"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "past_kv_flash_prefill_nosg", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "params_26"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "new_kv_past_flash_prefill_nosg", "priority": 34, "when": ["newKvPastOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Merge", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "params_11"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "params_28"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_rotary_flash_prefill_nosg", "priority": 36, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "standardSoftmax", "present.cosCacheT and rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "not present.headSinkT", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "hasRotary": true, "fusedRotary": true, "half": "gqaHeadDim / 2", "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "cos_cache", "sin_cache", "output_5", "params_26"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_softcap_flash_prefill_nosg", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "noAuxAttentionInputs", "attrs.softcap > 0", "attrs.smooth_softmax != 1", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasSoftcap": true, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "params_29"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_headsink_flash_prefill_nosg", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "present.headSinkT", "standardSoftmax", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasHeadSink": true, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "head_sink", "output_5", "params_26"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "past_kv_bias_headsink_flash_prefill_nosg", "priority": 34, "when": ["sharedKvFloatOk", "prefillFlashShapeOk", "cachedMaskClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "present.headSinkT", "standardSoftmax", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasHeadSink": true, "hasMask": true, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"copy\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "attn_mask", "head_sink", "output_5", "params_25"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "window_shift_flash_prefill_nosg", "priority": 34, "when": ["windowShiftOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"window_shift\"", "useSeqlens": true, "zeroScalar": "gqaScalar", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.WindowShift", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "seqlens_k", "params_28"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "qkv_present_flash_q32_broadcast", "priority": 30, "when": ["subgroupsWave32", "device.features.has(\"shader-f16\")", "tensorDtypes.queryT == \"float16\"", "flashShapeOk", "headDimPlan % 32 == 0", "headDimPlan >= 64", "headDimPlan <= 256", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "attrs.local_window_size <= 0"], "requires": { "features": ["subgroups"] }, "derive": { "hasBias": false, "hasCausal": false, "usesF16": true, "scalar": "\"f16\"", "inputVec4": "\"vec4\"", "outputScalar": "\"f16\"", "presentScalar": "qkvCacheScalar", "inputElement": "\"vec4\"", "outputElement": "\"vec4\"", "presentElement": "qkvCacheVec4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "kvHidden": "dim(shapes.keyT, 2)", "kvHiddenV4": "dim(shapes.keyT, 2) / 4", "presentVec4": true, "presentElem": "qkvCacheVec4", "kStep": 64, "qkGroups": 16 }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.FlashQ32Broadcast", "shader": "attn-flash-q32-broadcast.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"" }, "bindings": [ "query_6", "key_2", "value_2", "output_4", { "name": "params", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } ] } ], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), 32)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": 32 }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "qkv_present_flash_q32_shared", "priority": 29, "when": ["tensorDtypes.queryT == \"float16\"", "flashShapeOk", "headDimPlan % 4 == 0", "headDimPlan >= 64", "headDimPlan <= 256", "dim(shapes.queryT, 1) >= tunables.QKV_PREFILL_MIN_QUERY_TOKENS", "attrs.local_window_size <= 0", "ceilDiv(dim(shapes.queryT, 1), 64) <= min(device.limits.maxComputeWorkgroupsPerDimension, 65535)", "(headDimPlan / 4) * 32 * 16 <= device.limits.maxComputeWorkgroupStorageSize"], "requires": { "features": ["shader-f16"] }, "derive": { "hasBias": false, "hasCausal": false, "usesF16": true, "scalar": "\"f16\"", "inputVec4": "\"vec4\"", "outputScalar": "\"f16\"", "presentScalar": "qkvCacheScalar", "inputElement": "\"vec4\"", "outputElement": "\"vec4\"", "presentElement": "qkvCacheVec4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "dim(shapes.queryT, 2) / attrs.num_heads", "headDimV4": "(dim(shapes.queryT, 2) / attrs.num_heads) / 4", "kvHidden": "dim(shapes.keyT, 2)", "kvHiddenV4": "dim(shapes.keyT, 2) / 4", "presentVec4": true, "presentElem": "qkvCacheVec4", "kStep": 32, "qkGroups": 8, "qStep": 64 }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.FlashQ32Shared", "shader": "attn-flash-q32-broadcast.wgsl.jinja", "derive": { "usesF16": "tensorDtypes.queryT == \"float16\"", "layout": "\"bsh\"", "useSubgroups": "false" }, "bindings": [ "query_6", "key_2", "value_2", "output_4", { "name": "params", "struct": [ { "name": "qSeq", "type": "u32", "value": "dim(shapes.queryT, 1)" }, { "name": "kvSeq", "type": "u32", "value": "dim(shapes.keyT, 1)" }, { "name": "scale", "type": "f32", "value": "attrs.scale if attrs.scale else 0" } ] } ], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), 64)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["key_2", "value_2", "present_key", "present_value", "params"], "dispatch": { "x": "min(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((numel(shapes.presentKeyT) / 4), (copyWorkgroupSize)), 65535)", "z": 1 } } ] }, { "id": "past_kv", "priority": 30, "when": ["sharedKvFloatOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "gqaHeadDim % 2 == 0"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "mode": "\"copy\"", "usesF16": "tensorDtypes.queryT == \"float16\"", "inputScalar": "gqaScalar" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query_3", "kcache_3", "vcache_3", "output_3", "seqlens_k", "params_31"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] } ] }, { "id": "past_kv_rotary", "priority": 31, "when": ["sharedKvFloatOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "present.cosCacheT and rotaryRequested", "not present.qNormWeightT"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "half": "gqaHeadDim / 2", "hasRotary": true, "mode": "\"copy\"", "usesF16": "tensorDtypes.queryT == \"float16\"", "inputScalar": "gqaScalar", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query_3", "kcache_3", "vcache_3", "cos_cache", "sin_cache", "output_3", "seqlens_k", "params_31"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] } ] }, { "id": "past_kv_qnorm_rotary", "priority": 32, "when": ["sharedKvFloatOk", "gqaScalarDispatchOk", "gqaHeadDim % 2 == 0", "present.cosCacheT and rotaryRequested", "present.qNormWeightT"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "half": "gqaHeadDim / 2", "qkEps": "attrs.qk_norm_epsilon", "hasRotary": true, "hasQNorm": true, "mode": "\"copy\"", "usesF16": "tensorDtypes.queryT == \"float16\"", "inputScalar": "gqaScalar", "cosScalar": "\"f16\" if tensorDtypes.cosCacheT == \"float16\" else \"f32\"", "qNormScalar": "\"f16\" if tensorDtypes.qNormWeightT == \"float16\" else \"f32\"" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query_3", "kcache_3", "vcache_3", "cos_cache", "sin_cache", "q_norm_weight", "output_3", "seqlens_k", "params_31"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] } ] }, { "id": "new_kv_past", "priority": 30, "when": ["newKvPastOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "gqaHeadDim % 2 == 0"], "derive": { "headDim": "gqaHeadDim", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "mode": "\"merge\"", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "usesF16": "tensorDtypes.queryT == \"float16\"", "inputScalar": "gqaScalar" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Merge", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "params_11"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query_3", "kcache", "vcache", "output_3", "params_16"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "window_shift_append", "priority": 30, "when": ["windowShiftOk", "gqaScalarDispatchOk", "noAuxAttentionInputs", "gqaHeadDim % 2 == 0"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "mode": "\"window_shift\"", "usesF16": "tensorDtypes.queryT == \"float16\"", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar" }, "passes": [ { "id": "present", "name": "GroupQueryAttention.WindowShift", "shader": "gqa-present.wgsl.jinja", "bindings": ["past_k", "past_v", "new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query_3", "kcache", "vcache", "output_3", "seqlens_k", "params_16"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "new_kv_qnorm_rotary", "priority": 33, "when": ["copyWorkgroupOk", "gqaScalarDispatchOk", "present.seqlensKT", "not present.pastKeyT", "dim(shapes.keyT, 1) > 0", "tensorDtypes.queryT == \"float32\"", "not present.kScaleT", "present.cosCacheT and rotaryRequested", "present.qNormWeightT", "dim(shapes.queryT, 2) % attrs.num_heads == 0", "gqaHeadDim % 2 == 0"], "derive": { "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "half": "gqaHeadDim / 2", "qkEps": "attrs.qk_norm_epsilon", "hasRotary": true, "hasQNorm": true, "mode": "\"build\"", "hasKNorm": true }, "passes": [ { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k_2", "src_v_2", "cos_cache_2", "sin_cache_2", "k_norm_weight", "present_key_5", "present_value_5", "params_32"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query", "kcache_4", "vcache_4", "cos_cache_2", "sin_cache_2", "q_norm_weight_2", "output", "params_16"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } } ] }, { "id": "past_kv_bias", "priority": 30, "when": ["sharedKvCacheOk", "gqaScalarDispatchOk", "tensorDtypes.queryT == \"float32\"", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "not present.headSinkT"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "hasBias": true, "mode": "\"copy\"", "inputScalar": "gqaScalar" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query", "kcache_5", "vcache_5", "attn_bias", "output", "seqlens_k", "params_33"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] } ] }, { "id": "past_kv_headsink", "priority": 30, "when": ["sharedKvCacheOk", "gqaScalarDispatchOk", "tensorDtypes.queryT == \"float32\"", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "not present.attentionBiasT", "present.headSinkT"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "hasHeadSink": true, "mode": "\"copy\"", "inputScalar": "gqaScalar" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query", "kcache_5", "vcache_5", "head_sink", "output", "seqlens_k", "params_34"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] } ] }, { "id": "past_kv_bias_headsink", "priority": 30, "when": ["sharedKvCacheOk", "gqaScalarDispatchOk", "tensorDtypes.queryT == \"float32\"", "gqaHeadDim % 2 == 0", "not present.cosCacheT and not rotaryRequested", "not present.qNormWeightT", "present.attentionBiasT", "present.headSinkT"], "derive": { "useSeqlens": true, "headDim": "gqaHeadDim", "cooperative": "gqaCoop", "dispatchUnits": "gqaDispatchUnits", "qHeads": "attrs.num_heads", "kvHeads": "attrs.kv_num_heads", "qHidden": "dim(shapes.queryT, 2)", "packed": "gqaHeadDim", "hasBias": true, "hasHeadSink": true, "mode": "\"copy\"", "inputScalar": "gqaScalar" }, "passes": [ { "id": "attention", "name": "GroupQueryAttention.Attn", "shader": "gqa-attention.wgsl.jinja", "bindings": ["query", "kcache_5", "vcache_5", "attn_bias", "head_sink", "output", "seqlens_k", "params_35"], "dispatch": { "x": "min(dispatchUnits, 65535)", "y": "ceilDiv(dispatchUnits, 65535)", "z": 1 } }, { "id": "present", "name": "GroupQueryAttention.Present", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k", "src_v", "present_key_2", "present_value_2", "params_5"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.pastKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [{ "input": "src_k", "output": "present_key" }, { "input": "src_v", "output": "present_value" }] } ] }, { "id": "quant_int8_flash_prefill", "priority": 37, "when": ["quantizedPromptOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "attrs.kv_cache_bit_width == 8", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": false, "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"", "quantCacheFormat": "\"int8\"", "quantCacheElement": "\"i32\"", "quantCacheReadElement": "\"vec4\"", "packed": "gqaHeadDim", "cooperative": false, "dispatchUnits": 1, "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "mode": "\"build_quant\"", "bits": 8, "qmax": 127, "qmin": -128, "quantized": true }, "passes": [ { "id": "present", "name": "GroupQueryAttention.QuantPresent", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k_2", "src_v_2", "k_scale", "v_scale", "present_key_3", "present_value_3", "params_20"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.QuantFlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_8", "value_8", "k_scale", "v_scale", "output_5", "params_36"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "quant_int4_flash_prefill", "priority": 37, "when": ["quantizedPromptOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "attrs.kv_cache_bit_width == 4", "present.kScaleT", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": false, "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"", "quantCacheFormat": "\"int4\"", "quantCacheElement": "\"u32\"", "quantCacheReadElement": "\"u32\"", "packed": "gqaHeadDim / 2", "cooperative": false, "dispatchUnits": 1, "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "mode": "\"build_quant\"", "bits": 4, "qmax": 7, "qmin": -8, "quantized": true }, "passes": [ { "id": "present", "name": "GroupQueryAttention.QuantPresent", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k_2", "src_v_2", "k_scale", "v_scale", "present_key_3", "present_value_3", "params_20"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.QuantFlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_8", "value_8", "k_scale", "v_scale", "output_5", "params_36"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "quant_int8_flash_prefill_nosg", "priority": 37, "when": ["quantizedPromptOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "attrs.kv_cache_bit_width == 8", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": false, "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"", "quantCacheFormat": "\"int8\"", "quantCacheElement": "\"i32\"", "quantCacheReadElement": "\"vec4\"", "packed": "gqaHeadDim", "cooperative": false, "dispatchUnits": 1, "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "mode": "\"build_quant\"", "bits": 8, "qmax": 127, "qmin": -128, "quantized": true }, "passes": [ { "id": "present", "name": "GroupQueryAttention.QuantPresent", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k_2", "src_v_2", "k_scale", "v_scale", "present_key_3", "present_value_3", "params_20"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.QuantFlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_8", "value_8", "k_scale", "v_scale", "output_5", "params_36"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "quant_int4_flash_prefill_nosg", "priority": 37, "when": ["quantizedPromptOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "attrs.kv_cache_bit_width == 4", "present.kScaleT", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": false, "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"", "quantCacheFormat": "\"int4\"", "quantCacheElement": "\"u32\"", "quantCacheReadElement": "\"u32\"", "packed": "gqaHeadDim / 2", "cooperative": false, "dispatchUnits": 1, "qHeads": "attrs.num_heads", "qHidden": "dim(shapes.queryT, 2)", "mode": "\"build_quant\"", "bits": 4, "qmax": 7, "qmin": -8, "quantized": true }, "passes": [ { "id": "present", "name": "GroupQueryAttention.QuantPresent", "shader": "gqa-present.wgsl.jinja", "bindings": ["src_k_2", "src_v_2", "k_scale", "v_scale", "present_key_3", "present_value_3", "params_20"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.QuantFlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_8", "value_8", "k_scale", "v_scale", "output_5", "params_36"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), cachedPrefillQueryTile)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "share_append_split_decode_splitk", "priority": 39, "when": ["shareAppendOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"\" == \"\" else true"], "derive": { "cacheSeqlens": "true", "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge_share\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"\" == \"\"", "useSubgroups": "\"\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "seqlens_k", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "share_append_split_decode_splitk_nosg", "priority": 39, "when": ["shareAppendOk", "decodeFlashShapeOk", "cachedSplitScratchFits", "plainAttentionOptions", "subgroupSplitK if \"_nosg\" == \"\" else true"], "derive": { "cacheSeqlens": "true", "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "numSplits": "cachedNumSplits", "hasWindow": "attrs.local_window_size > 0", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge_share\"", "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "zeroScalar": "gqaScalar", "combineSubgroups": "\"_nosg\" == \"\"", "useSubgroups": "\"_nosg\" == \"\"" }, "intermediates": [ { "id": "partialOut", "dtype": "float32", "shape": "[dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits) * gqaHeadDim]" }, { "id": "partialStats", "dtype": "float32", "shape": "[2 * dim(shapes.queryT, 0) * attrs.num_heads * (cachedNumSplits)]" } ], "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "split_attention", "name": "GroupQueryAttention.DecodeSplitK", "shader": "attn-flash-decode-splitk.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "partial_out", "partial_stats", "seqlens_k", "params_27"], "dispatch": { "x": "cachedNumSplits", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } }, { "id": "merge", "name": "GroupQueryAttention.DecodeSplitKMerge", "shader": "attn-flash-decode-splitk-merge.wgsl.jinja", "derive": { "layout": "\"bhsd\"" }, "bindings": ["partial_out_2", "partial_stats_2", "output_5"], "dispatch": { "x": 1, "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] }, { "id": "share_append_split_flash_prefill", "priority": 35, "when": ["shareAppendOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedClusterTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge_share\"", "useSeqlens": true, "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"\" == \"_nosg\"", "useSubgroups": "\"\" == \"\"" }, "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "seqlens_k", "params_28"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" }, "subgroupCollectivesWidth": "portable" } ] }, { "id": "share_append_split_flash_prefill_nosg", "priority": 35, "when": ["shareAppendOk", "prefillFlashShapeOk", "cachedClusterStorageOk if \"_nosg\" == \"\" else cachedNoSgClusterStorageOk", "plainAttentionOptions", "(subgroupCachedPrefillCluster and cachedF32ClusterRegisterGeometry) if \"_nosg\" == \"\" else true"], "derive": { "scalar": "gqaScalar", "usesF16": "tensorDtypes.queryT == \"float16\"", "hasCausal": true, "causalRightAlign": true, "hasBias": false, "hasMask": false, "maskIsBool": false, "hasWindow": "attrs.local_window_size > 0", "headDim": "gqaHeadDim", "headDimV4": "gqaHeadDim / 4", "qHiddenV4": "dim(shapes.queryT, 2) / 4", "qNumHeads": "attrs.num_heads", "kvNumHeads": "attrs.kv_num_heads", "TILE_Q": "cachedPrefillQueryTile", "TILE_K": "cachedNoSgTileK", "LPQ": "cachedPrefillLanesPerQuery", "kvHeads": "attrs.kv_num_heads", "packed": "gqaHeadDim", "mode": "\"merge_share\"", "useSeqlens": true, "inputVec4": "gqaInputVec4", "inputScalar": "gqaScalar", "batchNoSgReduction": "\"_nosg\" == \"_nosg\"", "useSubgroups": "\"_nosg\" == \"\"" }, "passes": [ { "id": "present_retain", "name": "GroupQueryAttention.MergeShareRetain", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"retain\"" }, "bindings": ["past_k", "past_v", "present_key_2", "present_value_2", "seqlens_k", "params_12"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.presentKeyT, 2)), (copyWorkgroupSize)), 65535)", "z": 1 }, "viewAlias": [ { "input": "past_k", "output": "present_key" }, { "input": "past_v", "output": "present_value" } ] }, { "id": "present_append", "name": "GroupQueryAttention.MergeShareAppend", "shader": "gqa-present.wgsl.jinja", "derive": { "shareRegion": "\"append\"" }, "bindings": ["new_k", "new_v", "present_key_2", "present_value_2", "seqlens_k", "params_15"], "dispatch": { "x": "min(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "y": "ceilDiv(ceilDiv((dim(shapes.queryT, 0) * attrs.kv_num_heads * dim(shapes.keyT, 1)), (copyWorkgroupSize)), 65535)", "z": 1 } }, { "id": "attention", "name": "GroupQueryAttention.FlashPrefill", "shader": "attn-flash-prefill-cluster.wgsl.jinja", "derive": { "layout": "\"bsh\"", "qLayout": "\"bsh\"", "kvLayout": "\"bhsd\"" }, "bindings": ["query_8", "key_7", "value_7", "output_5", "seqlens_k", "params_28"], "dispatch": { "x": "ceilDiv(dim(shapes.queryT, 1), TILE_Q)", "y": "attrs.num_heads", "z": "dim(shapes.queryT, 0)" } } ] } ] }