Download build/webgpu/test.json from webgpu-kernels/com.microsoft.MatMulNBits: direct link, hf CLI and curl.
- Browser
- Download file 102 kB
-
https://huggingface.co/kernels/webgpu-kernels/com.microsoft.MatMulNBits/resolve/v1/build/webgpu/test.json
- Command line
-
hf download hf://webgpu-kernels/com.microsoft.MatMulNBits@v1/build/webgpu/test.json
-
curl -L -o test.json https://huggingface.co/kernels/webgpu-kernels/com.microsoft.MatMulNBits/resolve/v1/build/webgpu/test.json
102 kB
| { | |
| "fixtureArrays": { | |
| "q4_prefill_tiled_reg_tailk_m32_k33_n4096_input_bT": [16, 50, 84, 118, 152, 186, 220, 254, 135, 119, 102, 85, 68, 51, 34, 17], | |
| "q4_weight_cycle_b_t": [16, 50, 84, 118, 152, 186, 220, 254, 33, 67, 101, 135, 169, 203, 237, 15], | |
| "quant_scale_cycle_t": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07, 0.085], | |
| "mixed_weight_cycle_b_t": [17, 200, 91, 45, 233, 128, 7, 176, 250, 33, 142, 99, 210, 64, 188, 121], | |
| "q8_zero_bias_gemv_m1_tail_n5_input_bT": [19, 56, 93, 130, 167, 204, 241, 22, 59, 96, 133, 170, 207, 244, 25, 62], | |
| "ort_f16_large_k_accumulator_cancellation_m1_k8192_n8_input_scalesT": [16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16, -16] | |
| }, | |
| "cases": [ | |
| { | |
| "name": "q4_zero_bias_prefill_tile4x4_large", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [6.0, 7.0, 8.0, 9.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 64, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_zero_bias_prefill_tile4x4_small_m8", | |
| "provenance": { | |
| "notes": "M=8 rows of 4-bit quantized MatMulNBits with K=128, N=64, block_size=32, explicit zero points [6,7,8,9], and bias check correct dequantization and bias addition at a small row count." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [6.0, 7.0, 8.0, 9.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_tile4x4_small_m8", | |
| "provenance": { | |
| "notes": "An eight-row q8 prefill with zero points and bias exercises q8 unpacking and the bias epilogue on the small-M tile4x4 route." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "ort_projection_q4_block16_tail_zero_bias", | |
| "provenance": { | |
| "source": "onnxruntime/test/contrib_ops/matmul_4bits_test.cc", | |
| "test": "MatMulNBits.Float32_4b_Accuracy0", | |
| "notes": "Rank-2 projection of ORT q4 block-size coverage with per-block zero points and bias." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [2, 17], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.19, "cosStep": 0.37 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [2, 2, 8], | |
| "data": { | |
| "kind": "values", | |
| "values": [16, 50, 84, 118, 152, 186, 220, 254, 33, 0, 0, 0, 0, 0, 0, 0, 255, 238, 221, 204, 187, 170, 153, 136, 240, 0, 0, 0, 0, 0, 0, 0] | |
| } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [2, 2], | |
| "data": { "kind": "values", "values": [0.04, 0.08, 0.06, 0.12] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [2, 2], | |
| "data": { "kind": "values", "values": [7.0, 1.0, 8.0, 0.0] } | |
| }, | |
| "biasT": { "dtype": "float32", "shape": [2], "data": { "kind": "values", "values": [0.5, -1.0] } } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [2, 2], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 17, "N": 2, "bits": 4, "block_size": 16 } | |
| }, | |
| { | |
| "name": "ort_projection_q4_block32_tail_no_zero", | |
| "provenance": { | |
| "source": "onnxruntime/test/contrib_ops/matmul_4bits_test.cc", | |
| "test": "MatMulNBits.Float32_4b_Accuracy0", | |
| "notes": "Rank-2 deterministic projection of ORT q4 coverage with blockSize=32 and tail K." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [2, 33], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.75 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [2, 2, 16], | |
| "data": { | |
| "kind": "values", | |
| "values": [16, 50, 84, 118, 152, 186, 220, 254, 16, 50, 84, 118, 152, 186, 220, 254, 135, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 255, 238, 221, 204, 187, 170, 153, 136, 119, 102, 85, 68, 51, 34, 17, 0, 248, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] | |
| } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [2, 2], | |
| "data": { "kind": "values", "values": [0.03, 0.06, 0.04, 0.07] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [2, 2], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 33, "N": 2, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_block512_schema_legal_cpu_provider_gap_exact", | |
| "provenance": { | |
| "source": "onnxruntime/test/contrib_ops/matmul_4bits_test.cc", | |
| "test": "MatMulNBits.UnsupportedBlockSize_512", | |
| "notes": "The schema permits power-of-two block sizes of at least 16, including 512. With A=1, q4 code 9, default zero point 8, and scale 0.25, the result is exactly `512 * (9 - 8) * 0.25 = 128`." | |
| }, | |
| "inputs": { | |
| "aT": { "dtype": "float32", "shape": [1, 512], "data": { "kind": "constant", "value": 1.0 } }, | |
| "bT": { "dtype": "uint8", "shape": [1, 1, 256], "data": { "kind": "constant", "value": 153 } }, | |
| "scalesT": { "dtype": "float32", "shape": [1, 1], "data": { "kind": "constant", "value": 0.25 } } | |
| }, | |
| "outputs": { | |
| "yT": { "dtype": "float32", "shape": [1, 1], "tolerance": 0, "data": { "kind": "values", "values": [128.0] } } | |
| }, | |
| "attrs": { "K": 512, "N": 1, "bits": 4, "block_size": 512 } | |
| }, | |
| { | |
| "name": "q4_no_zero_prefill_sgmat_large", | |
| "inputs": { | |
| "aT": { | |
| "shape": [128, 128], | |
| "dtype": "float32", | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "shape": [64, 4, 16], | |
| "dtype": "uint8", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "shape": [64, 4], | |
| "dtype": "float32", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_no_zero_prefill_sgmat_m32_storage_floor", | |
| "provenance": { | |
| "notes": "The 32-row subgroup-matrix tile fits the WebGPU minimum 16 KiB workgroup-storage limit." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 32, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_zero_bias_prefill_sgmat_m32_storage_floor", | |
| "provenance": { | |
| "notes": "Explicit zero points and bias exercise the 32-row subgroup-matrix tile at the 16 KiB workgroup-storage floor." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [7.0, 8.0, 6.0, 9.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 32, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_no_zero_prefill_g64_sgmat", | |
| "provenance": { | |
| "notes": "A compact q4 prefill with block size 64 and the default zero point exercises the subgroup-matrix route and its portable tiled fallback." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "shape": [64, 128], | |
| "dtype": "float32", | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "shape": [64, 2, 32], | |
| "dtype": "uint8", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "shape": [64, 2], | |
| "dtype": "float32", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [64, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 64 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_sgmat_large", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_sgmat_m32_storage_floor", | |
| "provenance": { | |
| "notes": "A q8 prefill exercises the 32-row subgroup-matrix tile at the 16 KiB workgroup-storage floor." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 1, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 32, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_sgmat_large", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_reg_vec4_splitk_m128_k1024_n1024", | |
| "provenance": { | |
| "notes": "Without zero points or bias, 32 output tiles over K=1024 use two K slices whose float32 partial sums are combined." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 32, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 1024], "tolerance": 0.005 } }, | |
| "attrs": { "K": 1024, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_reg_vec4_splitk_m128_k1024_n1024", | |
| "provenance": { | |
| "notes": "Zero points affect the complete reduction, and bias is added exactly once to the final output." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 32, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [1024, 32], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 1024], "tolerance": 0.005 } }, | |
| "attrs": { "K": 1024, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_only_prefill_reg_vec4_splitk_m128_k1024_n1024", | |
| "provenance": { "notes": "Zero points affect the complete reduction without an output bias." }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 32, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [1024, 32], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 1024], "tolerance": 0.005 } }, | |
| "attrs": { "K": 1024, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_bias_only_prefill_reg_vec4_splitk_m128_k1024_n1024", | |
| "provenance": { | |
| "notes": "With bias and the implicit zero point, the split-K four-wide route adds bias once after combining the partial sums." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 32, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 1024], "tolerance": 0.005 } }, | |
| "attrs": { "K": 1024, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_sgmat_m32_storage_floor", | |
| "provenance": { | |
| "notes": "Q8 weights with explicit zero points and bias exercise the 32-row subgroup-matrix tile at the 16 KiB storage floor." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 1, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 32, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_tile4x4_unaligned", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [12, 2, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [12, 2], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 12], "tolerance": 0.001 } }, | |
| "attrs": { "K": 64, "N": 12, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_f16_zero_points_prefill_sgmat", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float16", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float16", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [7.0, 8.0, 9.0, 6.0] } | |
| }, | |
| "biasT": { "dtype": "float16", "shape": [64], "data": { "kind": "constant", "value": 0.0 } } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [128, 64], "tolerance": 0.05 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_no_zero_prefill_sgmat_unaligned_n72", | |
| "inputs": { | |
| "aT": { | |
| "shape": [128, 128], | |
| "dtype": "float32", | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "shape": [72, 4, 16], | |
| "dtype": "uint8", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "shape": [72, 4], | |
| "dtype": "float32", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 72], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 72, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_sgmat_unaligned_n72", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [72, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [72, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 72], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 72, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_sgmat_unaligned_n72_m32", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [72, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [72, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 72], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 72, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_sgmat_unaligned_n72", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [72, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [72, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [72, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [72], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 72], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 72, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_sgmat_unaligned_n65_m33", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [33, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [65, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [65, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [65, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [65], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [33, 65], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 65, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_sgmat_unaligned_n72_m32_bs16", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [72, 8, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [72, 8], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [72, 8], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [72], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 72], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 72, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "empty_zero_dim", | |
| "provenance": { | |
| "notes": "Zero-length M (rows) axis of A (empty input). With M=0 the N/K-driven packed uint8 B[4,1,16] and scales[4,1] stay normal and Y=[0,N] is empty. K=32,N=4,bits=4,block_size=32." | |
| }, | |
| "inputs": { | |
| "aT": { "dtype": "float32", "shape": [0, 32], "data": { "kind": "values", "values": [] } }, | |
| "bT": { "dtype": "uint8", "shape": [4, 1, 16], "data": { "kind": "constant", "value": 17 } }, | |
| "scalesT": { "dtype": "float32", "shape": [4, 1], "data": { "kind": "constant", "value": 0.05 } } | |
| }, | |
| "outputs": { | |
| "yT": { "dtype": "float32", "shape": [0, 4], "tolerance": 0, "data": { "kind": "values", "values": [] } } | |
| }, | |
| "attrs": { "K": 32, "N": 4, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "empty_zero_dim_f16", | |
| "provenance": { "notes": "f16 variant of the empty zero-M-axis case." }, | |
| "inputs": { | |
| "aT": { "dtype": "float16", "shape": [0, 32], "data": { "kind": "values", "values": [] } }, | |
| "bT": { "dtype": "uint8", "shape": [4, 1, 16], "data": { "kind": "constant", "value": 17 } }, | |
| "scalesT": { "dtype": "float16", "shape": [4, 1], "data": { "kind": "constant", "value": 0.05 } } | |
| }, | |
| "outputs": { | |
| "yT": { "dtype": "float16", "shape": [0, 4], "tolerance": 0, "data": { "kind": "values", "values": [] } } | |
| }, | |
| "attrs": { "K": 32, "N": 4, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_no_zero_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "shape": [512, 128], | |
| "dtype": "float32", | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "shape": [1024, 4, 16], | |
| "dtype": "uint8", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "shape": [1024, 4], | |
| "dtype": "float32", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_f16_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "shape": [512, 128], | |
| "dtype": "float16", | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "shape": [1024, 4, 16], | |
| "dtype": "uint8", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "shape": [1024, 4], | |
| "dtype": "float16", | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [512, 1024], "tolerance": 0.0001, "relTolerance": 0.001 } }, | |
| "provenance": { "notes": "Float16 activations exercise float16 A/B staging on the register-tiled prefill path." }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_zero_bias_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": [6.0, 7.0, 8.0, 9.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_f16_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float16", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [512, 1024], "tolerance": 0.0001, "relTolerance": 0.001 } }, | |
| "provenance": { | |
| "notes": "Eight-bit weights give the widest dequantized range, exercising float16 staging and store rounding on the register-tiled prefill path." | |
| }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_no_zero_prefill_tile4x4_partial_row_tile_m6", | |
| "provenance": { | |
| "notes": "M=6 rows of 4-bit MatMulNBits (K=64, N=8, block_size=32, default zero point, no bias) check a row count that is not a multiple of four." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [6, 64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [8, 2, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [8, 2], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [6, 8], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 64, "N": 8, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_gemv_default_zero_m1_tail_n7", | |
| "provenance": { | |
| "notes": "M=1 and N=7 leave three live columns in the final four-column output group. The guarded stores must write those columns without accessing the eighth column; zero points are omitted." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [7, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [7, 1], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 7], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 32, "N": 7, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_gemv_default_zero_m1_odd_k33_n5", | |
| "provenance": { | |
| "notes": "With K=33, the last stored byte of the second q4 block contains one live nibble and one beyond K. The GEMV must guard the out-of-range activation." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 33], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [5, 2, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [5, 2], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 5], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 33, "N": 5, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q2_gemv_default_zero_m1_k18_n5", | |
| "provenance": { | |
| "notes": "Four q2 codes share each byte, and K=18 leaves two codes in the final byte beyond K. The GEMV must unpack the live codes and guard both out-of-range activations." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 18], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [5, 2, 4], | |
| "data": { "kind": "cycle", "values": [27, 180, 75, 226, 33, 150, 201, 108, 57, 246, 129, 66, 195, 84, 237] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [5, 2], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 5], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 18, "N": 5, "bits": 2, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q4_gemv_default_zero_m1_n13_ncols8", | |
| "tunables": { "GEMV_N_COLS": 8 }, | |
| "provenance": { | |
| "notes": "N=13 with M=1, K=32, block_size=32, 4-bit weights and the default zero point checks a column count that is not a multiple of eight." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [13, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [13, 1], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 13], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 32, "N": 13, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_gemv_default_zero_m1_tail_n7_ncols8", | |
| "tunables": { "GEMV_N_COLS": 8 }, | |
| "provenance": { | |
| "notes": "N=7 with M=1, K=32, block_size=32, 4-bit weights and the default zero point checks that all seven output columns are dequantized and written correctly." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [7, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [7, 1], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 7], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 32, "N": 7, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_gemv_m1_tail_n5", | |
| "provenance": { | |
| "notes": "M=1 and N=5 leave one live column in the final four-column output group. The case combines guarded tail stores with q8 unpacking, explicit zero points, and the bias epilogue." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [5, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q8_zero_bias_gemv_m1_tail_n5_input_bT" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [5, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [5, 1], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [5], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 5], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 16, "N": 5, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q8_zero_only_gemv_m1_tail_n5", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "An M=1 q8 GEMV with explicit zero points and no bias isolates the zero-point binding layout and epilogue." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [5, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q8_zero_bias_gemv_m1_tail_n5_input_bT" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [5, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [5, 1], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 5], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 16, "N": 5, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q8_bias_only_gemv_m1_tail_n5", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "An M=1 q8 GEMV with bias and the schema-default zero point isolates the bias binding layout and epilogue." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [5, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q8_zero_bias_gemv_m1_tail_n5_input_bT" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [5, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [5], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 5], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 16, "N": 5, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_gemv_m1_tail_n5_ncols8", | |
| "tunables": { "GEMV_N_COLS": 8 }, | |
| "provenance": { | |
| "notes": "N=5 with M=1, K=16, block_size=16, 8-bit weights, explicit per-block zero points and a bias checks that all five output columns are dequantized and written correctly." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 16], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [5, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q8_zero_bias_gemv_m1_tail_n5_input_bT" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [5, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [5, 1], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [5], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 5], "tolerance": 0.00001 } }, | |
| "attrs": { "K": 16, "N": 5, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_naive_fallback_tailK_m3_n6", | |
| "provenance": { | |
| "notes": "K=17 with block size 16 leaves a partial quantization block. M=3 and N=6 check q8 unpacking, per-block zero points, bias, and the K tail." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [3, 17], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.75 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [6, 2, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q8_zero_bias_gemv_m1_tail_n5_input_bT" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [6, 2], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [6, 2], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [6], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [3, 6], "tolerance": 0.001 } }, | |
| "attrs": { "K": 17, "N": 6, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q4_prefill_tiled_reg_tailk_m32_k33_n4096", | |
| "provenance": { | |
| "notes": "A 33-element reduction leaves a partial final quantization block; padded weights must not contribute to the result or read past the activation input." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 33], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [4096, 2, 16], | |
| "data": { | |
| "kind": "cycle", | |
| "values": { "$ref": "#/fixtureArrays/q4_prefill_tiled_reg_tailk_m32_k33_n4096_input_bT" } | |
| } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [4096, 2], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 4096], "tolerance": 0.001 } }, | |
| "attrs": { "K": 33, "N": 4096, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_prefill_tiled_reg_aligned_m32_k32_n4096_at_workgroup_floor", | |
| "provenance": { | |
| "source": "synthetic", | |
| "notes": "Aligned K=32, M=32, and N=4096 produce exactly 64 workgroups, exercising the register-tiled prefill route at its workgroup floor." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [4096, 1, 16], | |
| "data": { | |
| "kind": "cycle", | |
| "values": { "$ref": "#/fixtureArrays/q4_prefill_tiled_reg_tailk_m32_k33_n4096_input_bT" } | |
| } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [4096, 1], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 4096], "tolerance": 0.001 } }, | |
| "attrs": { "K": 32, "N": 4096, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_prefill_tiled_reg_tailk_m32_k33_n512_below_wide_n_floor", | |
| "provenance": { | |
| "notes": "K=33 exercises a padded final block while N=512 places the register-tiled prefill route in its relaxed narrow-output band." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 33], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [512, 2, 16], | |
| "data": { | |
| "kind": "cycle", | |
| "values": { "$ref": "#/fixtureArrays/q4_prefill_tiled_reg_tailk_m32_k33_n4096_input_bT" } | |
| } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [512, 2], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 512], "tolerance": 0.001 } }, | |
| "attrs": { "K": 33, "N": 512, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_f16_prefill_reg_tailk_m32_k33_n4096", | |
| "provenance": { | |
| "notes": "f16 activations through the register-tiled general (non-block-aligned) loader and the M=32 portable geometry, plus the padded tail K." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float16", | |
| "shape": [32, 33], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [4096, 2, 16], | |
| "data": { | |
| "kind": "cycle", | |
| "values": { "$ref": "#/fixtureArrays/q4_prefill_tiled_reg_tailk_m32_k33_n4096_input_bT" } | |
| } | |
| }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [4096, 2], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [32, 4096], "tolerance": 0.0001, "relTolerance": 0.001 } }, | |
| "attrs": { "K": 33, "N": 4096, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_no_zero_prefill_odd_n_fallback", | |
| "provenance": { | |
| "notes": "An eight-row q8 projection with N=17 and no zero points or bias checks the final odd output column." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [17, 2, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [17, 2], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 17], "tolerance": 0.001 } }, | |
| "attrs": { "K": 64, "N": 17, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_zero_only_prefill_sgmat_m32_storage_floor", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is zeroPointsT; the expected output isolates explicit zero-point decoding from bias handling." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [7.0, 8.0, 6.0, 9.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 32, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_bias_only_prefill_sgmat_m32_storage_floor", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is biasT; zero-point decoding uses the schema default, isolating the bias epilogue." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 1, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 32, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_only_prefill_sgmat_large", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is zeroPointsT; the expected output isolates explicit zero-point decoding from bias handling." | |
| }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_bias_only_prefill_sgmat_large", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is biasT; zero-point decoding uses the schema default, isolating the bias epilogue." | |
| }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_zero_only_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": [6.0, 7.0, 8.0, 9.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.0001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is zeroPointsT; the expected output isolates explicit zero-point decoding from bias handling." | |
| }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_bias_only_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.0001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is biasT; zero-point decoding uses the schema default, isolating the bias epilogue." | |
| }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_only_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is zeroPointsT; the expected output isolates explicit zero-point decoding from bias handling." | |
| }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_bias_only_prefill_reg_m512_n1024", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is biasT; zero-point decoding uses the schema default, isolating the bias epilogue." | |
| }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_zero_only_prefill_tile4x4_small_m8", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is zeroPointsT; the expected output isolates explicit zero-point decoding from bias handling." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [6.0, 7.0, 8.0, 9.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_bias_only_prefill_tile4x4_small_m8", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is biasT; zero-point decoding uses the schema default, isolating the bias epilogue." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 64], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_only_prefill_tile4x4_small_m8", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is zeroPointsT; the expected output isolates explicit zero-point decoding from bias handling." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0, 126.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_bias_only_prefill_tile4x4_small_m8", | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is biasT; zero-point decoding uses the schema default, isolating the bias epilogue." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_zero_only_prefill_tile4x4_large", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [6.0, 7.0, 8.0, 9.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.0001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is zeroPointsT; the expected output isolates explicit zero-point decoding from bias handling." | |
| }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_bias_only_prefill_tile4x4_large", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.0001 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/matmul_nbits.cc", | |
| "test": "independent has_zero_points/has_bias epilogues", | |
| "notes": "The only optional input is biasT; zero-point decoding uses the schema default, isolating the bias epilogue." | |
| }, | |
| "attrs": { "K": 128, "N": 64, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_dp4a_prefill_al4_m64_k256_n64_b32", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [64, 256], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 8, 16], | |
| "data": { | |
| "kind": "cycle", | |
| "values": [37, 201, 116, 88, 250, 13, 164, 95, 142, 7, 233, 60, 178, 121, 44, 209, 86, 155] | |
| } | |
| }, | |
| "scalesT": { "dtype": "float32", "shape": [64, 8], "data": { "kind": "linspace", "start": 0.01, "end": 0.05 } } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [64, 64], "tolerance": 0.03 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/dp4a_matmul_nbits.cc", | |
| "test": "accuracy_level=4 int8 activation quantization (DP4AMatMulQuantize + DP4AMatMulNBits)", | |
| "notes": "Accuracy level 4 quantizes A per 128-element block using `max_abs / 127` and signed-normalized rounding, then multiplies q4 weights with packed integer dots. Tolerances account for the permitted activation-quantization error." | |
| }, | |
| "attrs": { "K": 256, "N": 64, "bits": 4, "block_size": 32, "accuracy_level": 4 } | |
| }, | |
| { | |
| "name": "q4_dp4a_prefill_al4_m48_k128_n16_b64", | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [48, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.027 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [16, 2, 32], | |
| "data": { "kind": "cycle", "values": [171, 205, 46, 251, 218, 201, 31, 186, 237, 220, 154] } | |
| }, | |
| "scalesT": { "dtype": "float32", "shape": [16, 2], "data": { "kind": "linspace", "start": 0.02, "end": 0.06 } } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [48, 16], "tolerance": 0.03 } }, | |
| "provenance": { | |
| "source": "onnxruntime/contrib_ops/webgpu/quantization/dp4a_matmul_nbits.cc", | |
| "test": "accuracy_level=4 int8 activation quantization (DP4AMatMulQuantize + DP4AMatMulNBits)", | |
| "notes": "Diverges from the upstream test's inputs (inputs.bT cycle [37, 201, 116, 88, 250, 13, 164, 95, 142, 7, 233] -> cycle [171, 205, 46, 251, 218, 201, 31, 186, 237, 220, 154]); the expected output is recomputed by the CPU reference for the new inputs. Accuracy level 4 quantizes A per 128-element block using `max_abs / 127` and signed-normalized rounding, then multiplies q4 weights with packed integer dots. Tolerances account for the permitted activation-quantization error, so the weight codes sit mostly above the default zero point: a near-zero-mean code cycle cancels the product down to the magnitude of that error, leaving no tolerance that can see a scale mistake." | |
| }, | |
| "attrs": { "K": 128, "N": 16, "bits": 4, "block_size": 64, "accuracy_level": 4 } | |
| }, | |
| { | |
| "name": "q2_zero_bias_prefill_tile4x4_small_m8", | |
| "provenance": { | |
| "notes": "2-bit weights pack four codes per stored byte, low codes first along K, with a default zero point of 2. Small-M prefill on the 4x4 output tile." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [8, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 8], | |
| "data": { "kind": "cycle", "values": [27, 180, 75, 226, 33, 150, 201, 108, 57, 246, 129, 66, 195, 84, 237] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [1.0, 2.0, 3.0, 0.0, 2.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [8, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 2, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q2_no_zero_prefill_sgmat_large", | |
| "provenance": { | |
| "notes": "2-bit weights pack four codes per stored byte, low codes first along K, with a default zero point of 2. Prefill wide enough for the subgroup-matrix tile." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [128, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 8], | |
| "data": { "kind": "cycle", "values": [27, 180, 75, 226, 33, 150, 201, 108, 57, 246, 129, 66, 195, 84, 237] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [128, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 2, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q2_zero_bias_prefill_sgmat_m32_storage_floor", | |
| "provenance": { | |
| "notes": "2-bit weights pack four codes per stored byte, low codes first along K, with a default zero point of 2. Subgroup-matrix tile at the M=32 shared-storage floor." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [32, 32], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 1, 8], | |
| "data": { "kind": "cycle", "values": [27, 180, 75, 226, 33, 150, 201, 108, 57, 246, 129, 66, 195, 84, 237] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 1], | |
| "data": { "kind": "cycle", "values": [1.0, 2.0, 3.0, 0.0, 2.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [32, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 32, "N": 64, "bits": 2, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q2_no_zero_prefill_reg_m512_n1024", | |
| "provenance": { | |
| "notes": "2-bit weights pack four codes per stored byte, low codes first along K, with a default zero point of 2. Register-tile prefill, whose loader unrolls by K and so needs its own unpack." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 8], | |
| "data": { "kind": "cycle", "values": [27, 180, 75, 226, 33, 150, 201, 108, 57, 246, 129, 66, 195, 84, 237] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 2, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q2_zero_bias_prefill_reg_m512_n1024", | |
| "provenance": { | |
| "notes": "2-bit weights pack four codes per stored byte, low codes first along K, with a default zero point of 2. Register-tile prefill with zero points and bias." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [512, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 4, 8], | |
| "data": { "kind": "cycle", "values": [27, 180, 75, 226, 33, 150, 201, 108, 57, 246, 129, 66, 195, 84, 237] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [1024, 4], | |
| "data": { "kind": "cycle", "values": [1.0, 2.0, 3.0, 0.0, 2.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [512, 1024], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 1024, "bits": 2, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_rank3_prefill_tile4x4", | |
| "provenance": { "notes": "Rank-3 A folding to eight rows, wide enough for the prefill output tile." }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [2, 4, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [64, 4, 32], | |
| "data": { "kind": "cycle", "values": [17, 200, 91, 45, 233, 128, 7, 176, 250, 33, 142, 99, 210, 64, 188] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [64, 4], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [2, 4, 64], "tolerance": 0.001 } }, | |
| "attrs": { "K": 128, "N": 64, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_only_naive_fallback_tailk_m3_n6", | |
| "provenance": { | |
| "notes": "K=17 with block_size=16 leaves one element in a partial final quantization block (M=3, N=6, 8-bit, explicit per-block zero points, no bias); checks that the partial block is weighted correctly without bias." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [3, 17], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.75 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [6, 2, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q8_zero_bias_gemv_m1_tail_n5_input_bT" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [6, 2], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [6, 2], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0] } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [3, 6], "tolerance": 0.001 } }, | |
| "attrs": { "K": 17, "N": 6, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q8_bias_only_naive_fallback_tailk_m3_n6", | |
| "provenance": { | |
| "notes": "K=17 with block_size=16 leaves one element in a partial final quantization block (M=3, N=6, 8-bit, default zero point, explicit bias); checks that the partial block and bias combine correctly." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [3, 17], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.75 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [6, 2, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q8_zero_bias_gemv_m1_tail_n5_input_bT" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [6, 2], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [6], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.2 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [3, 6], "tolerance": 0.001 } }, | |
| "attrs": { "K": 17, "N": 6, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q4_prefill_tile4x4_m4_k128_n4100_column_groups_partial_group", | |
| "provenance": { | |
| "notes": "K=128 at block 32 is sixteen packed words per column, so the small-M kernel runs 16 K-lanes and four column groups (16 columns) per workgroup. N=4100 is 256 full workgroups plus one whose second group starts at column 4096 and whose third and fourth groups lie past N, so the group-valid guard and the partial last group are both exercised at scale; M=4 is one row tile." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [4, 128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.35 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [4100, 4, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [4100, 4], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [4, 4100], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 128, "N": 4100, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_zero_bias_prefill_tile4x4_m5_k64_n300_column_groups", | |
| "provenance": { | |
| "notes": "q8 at K=64, block 32: sixteen packed words per column, four 8-bit codes per word, so a word carries four K-adjacent codes of one block. N=300 with sixteen columns per workgroup leaves a final workgroup whose last group holds columns 296..299 and M=5 leaves a second row tile with one live row, so the per-row and per-column guards run together with zero points and bias." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [5, 64], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.09, "cosStep": 0.27, "scale": 0.4 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [300, 2, 32], | |
| "data": { "kind": "cycle", "values": [17, 200, 91, 45, 233, 128, 7, 176, 250, 33, 142, 99, 210, 64, 188] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [300, 2], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [300, 2], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [300], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [5, 300], "tolerance": 0.001 } }, | |
| "attrs": { "K": 64, "N": 300, "bits": 8, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q4_gemv_default_zero_m1_k2080_n6_two_vector_trips", | |
| "provenance": { | |
| "notes": "K=2080 at block size 32 yields 65 sixteen-byte vectors per column, so lane 0 makes a second vector-loop trip and applies the 65th block's scale. N=6 leaves the second column group half empty." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 2080], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.031 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [6, 65, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [6, 65], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 6], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 2080, "N": 6, "bits": 4, "block_size": 32 } | |
| }, | |
| { | |
| "name": "q8_gemv_zero_bias_m1_k48_block16_word_vectors", | |
| "provenance": { | |
| "notes": "q8 at block 16 is a sixteen-byte blob, so the GEMV still loads whole sixteen-byte vectors (one per block) while q4 at block 16 would fall to single words; K=48 is three vectors and N=9 leaves a partial column group, with zero points and bias." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 48], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [9, 3, 16], | |
| "data": { "kind": "cycle", "values": [17, 200, 91, 45, 233, 128, 7, 176, 250, 33, 142, 99, 210, 64, 188] } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [9, 3], | |
| "data": { "kind": "cycle", "values": [0.04, 0.055, 0.05, 0.065, 0.06, 0.075, 0.07] } | |
| }, | |
| "zeroPointsT": { | |
| "dtype": "float32", | |
| "shape": [9, 3], | |
| "data": { "kind": "cycle", "values": [120.0, 132.0, 128.0, 140.0, 135.0, 118.0, 144.0] } | |
| }, | |
| "biasT": { | |
| "dtype": "float32", | |
| "shape": [9], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.1 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 9], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 48, "N": 9, "bits": 8, "block_size": 16 } | |
| }, | |
| { | |
| "name": "q4_gemv_default_zero_m1_block16_single_word_vectors_k40", | |
| "provenance": { | |
| "notes": "q4 at block 16 is an eight-byte blob, below the sixteen-byte vector, so the GEMV falls to one-word vectors (eight codes each) and K=40 ends half way through the third block: the last vector's trailing codes lie past K and must contribute zero." | |
| }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float32", | |
| "shape": [1, 40], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.19, "cosStep": 0.37 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [5, 3, 8], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float32", | |
| "shape": [5, 3], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float32", "shape": [1, 5], "tolerance": 0.0001 } }, | |
| "attrs": { "K": 40, "N": 5, "bits": 4, "block_size": 16 } | |
| }, | |
| { | |
| "name": "ort_f16_large_k_accumulator_cancellation_m1_k8192_n8", | |
| "provenance": { | |
| "notes": "Transcribed from ORT MatMulNBits.Float16_LargeK_AccumulatorOverflow (M=1 arm). A = 8 everywhere; the dequantized weight is +112 over the first half of K and -112 over the second, so the exact result is 0 while the running partial sum crosses the f16 ceiling (65504) at 114688 in any kernel that walks K in one accumulator. ORT puts the sign on B's codes (0xFF then 0x11); this package puts it on the block scale instead (+16 then -16) with every code 15, which yields a bit-identical dequantized weight matrix and lets B be one constant instead of a 4096-long cycle." | |
| }, | |
| "attrs": { "K": 8192, "N": 8, "bits": 4, "block_size": 32 }, | |
| "inputs": { | |
| "aT": { "dtype": "float16", "shape": [1, 8192], "data": { "kind": "constant", "value": 8.0 } }, | |
| "bT": { "dtype": "uint8", "shape": [8, 256, 16], "data": { "kind": "constant", "value": 255 } }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [8, 256], | |
| "data": { | |
| "kind": "cycle", | |
| "values": { "$ref": "#/fixtureArrays/ort_f16_large_k_accumulator_cancellation_m1_k8192_n8_input_scalesT" } | |
| } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [1, 8], "tolerance": 0.05 } } | |
| }, | |
| { | |
| "name": "ort_f16_large_k_accumulator_cancellation_m8_k8192_n8", | |
| "provenance": { | |
| "notes": "Transcribed from ORT MatMulNBits.Float16_LargeK_AccumulatorOverflow (M=8 arm). The same cancellation construction as the M=1 case checks float32 accumulation for eight output rows." | |
| }, | |
| "attrs": { "K": 8192, "N": 8, "bits": 4, "block_size": 32 }, | |
| "inputs": { | |
| "aT": { "dtype": "float16", "shape": [8, 8192], "data": { "kind": "constant", "value": 8.0 } }, | |
| "bT": { "dtype": "uint8", "shape": [8, 256, 16], "data": { "kind": "constant", "value": 255 } }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [8, 256], | |
| "data": { | |
| "kind": "cycle", | |
| "values": { "$ref": "#/fixtureArrays/ort_f16_large_k_accumulator_cancellation_m1_k8192_n8_input_scalesT" } | |
| } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [8, 8], "tolerance": 0.05 } } | |
| }, | |
| { | |
| "name": "f16_full_k_accumulator_cancellation_prefill_m64_k4096_n64", | |
| "provenance": { | |
| "notes": "Adapted from ORT MatMulNBits.Float16_LargeK_AccumulatorOverflow. With K=4,096, partial sums rise above the float16 finite ceiling (peak 1,835,008) before cancelling to an exact zero output." | |
| }, | |
| "attrs": { "K": 4096, "N": 64, "bits": 4, "block_size": 32 }, | |
| "inputs": { | |
| "aT": { "dtype": "float16", "shape": [64, 4096], "data": { "kind": "constant", "value": 8.0 } }, | |
| "bT": { "dtype": "uint8", "shape": [64, 128, 16], "data": { "kind": "constant", "value": 255 } }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [64, 128], | |
| "data": { | |
| "kind": "cycle", | |
| "values": [16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, 16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0, -16.0] | |
| } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.05 } } | |
| }, | |
| { | |
| "name": "ort_f16_decode_bias_m1_k1024_n128_b32", | |
| "provenance": { | |
| "notes": "Transcribed from ORT MatMulNBits.Float16_AccumulatorPrecisionOption_AllPaths, case {M=1, N=128, K=1024, block 32, accuracy_level 0, bias}. The provider option is not applicable here (accumulators are always f32); the shape is kept because it is the generic decode dispatch with a bias and there was no other float16 M=1 case. Weight/scale data reuse this file's existing fixture arrays." | |
| }, | |
| "attrs": { "K": 1024, "N": 128, "bits": 4, "block_size": 32 }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float16", | |
| "shape": [1, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.25 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [128, 32, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [128, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float16", | |
| "shape": [128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.125 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [1, 128], "tolerance": 0.01, "relTolerance": 0.005 } } | |
| }, | |
| { | |
| "name": "ort_f16_prefill_bias_m8_k1024_n128_b32", | |
| "provenance": { | |
| "notes": "Transcribed from ORT MatMulNBits.Float16_AccumulatorPrecisionOption_AllPaths, case {M=8, N=128, K=1024, block 32, accuracy_level 0, bias} (ORT's wide-tile arm)." | |
| }, | |
| "attrs": { "K": 1024, "N": 128, "bits": 4, "block_size": 32 }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float16", | |
| "shape": [8, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.25 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [128, 32, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [128, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "biasT": { | |
| "dtype": "float16", | |
| "shape": [128], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.41, "scale": 0.125 } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [8, 128], "tolerance": 0.01, "relTolerance": 0.005 } } | |
| }, | |
| { | |
| "name": "q2_block128_f16_zero1_prefill_m128_k1024_n1024", | |
| "provenance": { | |
| "notes": "M=128, K=1024, N=1024 with 2-bit weights in 128-value quantization blocks, a uniform explicit zero point of 1, and float16 operands check dequantization at a full-size projection shape." | |
| }, | |
| "attrs": { "K": 1024, "N": 1024, "bits": 2, "block_size": 128 }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float16", | |
| "shape": [128, 1024], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.25 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [1024, 8, 32], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/mixed_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [1024, 8], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| }, | |
| "zeroPointsT": { "dtype": "float16", "shape": [1024, 8], "data": { "kind": "constant", "value": 1.0 } } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [128, 1024], "tolerance": 0.01, "relTolerance": 0.01 } } | |
| }, | |
| { | |
| "name": "ort_f16_accuracy_level4_m8_k4096_n128_generic_route", | |
| "provenance": { | |
| "notes": "Transcribed from ORT MatMulNBits.Float16_AccumulatorPrecisionOption_AllPaths: M=8, N=128, K=4,096, block size 32, accuracy_level 4. The float16 inputs check float32 accumulation independently of provider-specific execution choices." | |
| }, | |
| "attrs": { "K": 4096, "N": 128, "bits": 4, "block_size": 32, "accuracy_level": 4 }, | |
| "inputs": { | |
| "aT": { | |
| "dtype": "float16", | |
| "shape": [8, 4096], | |
| "data": { "kind": "fillFloat32", "sinStep": 0.11, "cosStep": 0.23, "scale": 0.25 } | |
| }, | |
| "bT": { | |
| "dtype": "uint8", | |
| "shape": [128, 128, 16], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/q4_weight_cycle_b_t" } } | |
| }, | |
| "scalesT": { | |
| "dtype": "float16", | |
| "shape": [128, 128], | |
| "data": { "kind": "cycle", "values": { "$ref": "#/fixtureArrays/quant_scale_cycle_t" } } | |
| } | |
| }, | |
| "outputs": { "yT": { "dtype": "float16", "shape": [8, 128], "tolerance": 0.02, "relTolerance": 0.01 } } | |
| } | |
| ] | |
| } | |