Xenova's picture
Xenova HF Staff
sync 6fdf6301e2bb
5a12de3 verified
Raw History Blame
80.5 kB
{
"fixtureArrays": {
"ort_float32_broadcast_rank3_by_rank4_output_Y": [1, 3, 5, 33, 43, 53, 5, 23, 41, 85, 111, 137, 9, 43, 77, 137, 179, 221],
"ort_float32_rank3_by_rank2_output_Y": [20, 23, 26, 29, 56, 68, 80, 92, 92, 113, 134, 155, 128, 158, 188, 218],
"ort_float32_batched_rank4_input_A": [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15]
},
"cases": [
{
"name": "ort_float32_broadcast_rank4_by_rank3",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [3, 1, 1, 2],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0] }
},
"B": {
"dtype": "float32",
"shape": [2, 2, 2],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [3, 2, 1, 2],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [2.0, 3.0, 6.0, 7.0, 6.0, 11.0, 26.0, 31.0, 10.0, 19.0, 46.0, 55.0] }
}
}
},
{
"name": "ort_float32_broadcast_rank3_by_rank4",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 3, 2],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
},
"B": {
"dtype": "float32",
"shape": [3, 2, 2, 1],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [3, 2, 3, 1],
"tolerance": 0.000001,
"data": {
"kind": "values",
"values": { "$ref": "#/fixtureArrays/ort_float32_broadcast_rank3_by_rank4_output_Y" }
}
}
}
},
{
"name": "ort_float32_left_1d_batched_rhs",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": { "dtype": "float32", "shape": [2], "data": { "kind": "values", "values": [0.0, 1.0] } },
"B": {
"dtype": "float32",
"shape": [3, 2, 1],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [3, 1],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [1.0, 3.0, 5.0] }
}
}
},
{
"name": "ort_float32_right_1d_batched_lhs",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [3, 1, 2],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0] }
},
"B": { "dtype": "float32", "shape": [2], "data": { "kind": "values", "values": [0.0, 1.0] } }
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [3, 1],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [1.0, 3.0, 5.0] }
}
}
},
{
"name": "ort_float32_plain_2d",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [3, 4],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
},
"B": {
"dtype": "float32",
"shape": [4, 3],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [3, 3],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [42.0, 48.0, 54.0, 114.0, 136.0, 158.0, 186.0, 224.0, 262.0] }
}
}
},
{
"name": "ort_float32_rank3_by_rank2",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 2, 3],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
},
"B": {
"dtype": "float32",
"shape": [3, 4],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 2, 4],
"tolerance": 0.000001,
"data": { "kind": "values", "values": { "$ref": "#/fixtureArrays/ort_float32_rank3_by_rank2_output_Y" } }
}
}
},
{
"name": "ort_float32_rank3_by_broadcast_rank3",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 2, 3],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
},
"B": {
"dtype": "float32",
"shape": [1, 3, 4],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 2, 4],
"tolerance": 0.000001,
"data": { "kind": "values", "values": { "$ref": "#/fixtureArrays/ort_float32_rank3_by_rank2_output_Y" } }
}
}
},
{
"name": "ort_float32_singleton_rank3_by_rank3",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [1, 2, 3],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0] }
},
"B": {
"dtype": "float32",
"shape": [1, 3, 4],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [1, 2, 4],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [20.0, 23.0, 26.0, 29.0, 56.0, 68.0, 80.0, 92.0] }
}
}
},
{
"name": "ort_float32_batched_rank4",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 2, 2, 2],
"data": { "kind": "values", "values": { "$ref": "#/fixtureArrays/ort_float32_batched_rank4_input_A" } }
},
"B": {
"dtype": "float32",
"shape": [2, 2, 2, 2],
"data": { "kind": "values", "values": { "$ref": "#/fixtureArrays/ort_float32_batched_rank4_input_A" } }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 2, 2, 2],
"tolerance": 0.000001,
"data": {
"kind": "values",
"values": [2.0, 3.0, 6.0, 11.0, 46.0, 55.0, 66.0, 79.0, 154.0, 171.0, 190.0, 211.0, 326.0, 351.0, 378.0, 407.0]
}
}
}
},
{
"name": "ort_float32_broadcast_rank4_by_rank4",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": {
"dtype": "float32",
"shape": [1, 2, 3, 2],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
},
"B": {
"dtype": "float32",
"shape": [3, 2, 2, 1],
"data": { "kind": "values", "values": [0.0, 1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [3, 2, 3, 1],
"tolerance": 0.000001,
"data": {
"kind": "values",
"values": { "$ref": "#/fixtureArrays/ort_float32_broadcast_rank3_by_rank4_output_Y" }
}
}
}
},
{
"name": "ort_float32_vector_dot_scalar_output",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": { "dtype": "float32", "shape": [3], "data": { "kind": "values", "values": [0.0, 1.0, 2.0] } },
"B": { "dtype": "float32", "shape": [3], "data": { "kind": "values", "values": [0.0, 1.0, 2.0] } }
},
"outputs": {
"Y": { "dtype": "float32", "shape": [], "tolerance": 0.000001, "data": { "kind": "values", "values": [5.0] } }
}
},
{
"name": "ort_float32_alpha_zero_outputs_zero",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.DoubleTypeAlphaZero",
"notes": "Diverges from the upstream test's inputs (inputs.A values [1.0, 2.0, 3.0, 4.0] -> constant 2.0; inputs.B values [5.0, 6.0, 7.0, 8.0] -> constant 3.0); the expected output is recomputed by the CPU reference for the new inputs. A zero alpha scales the whole product away, so no operand value can reach the result and both operands are uniform fills."
},
"attrs": { "alpha": 0 },
"inputs": {
"A": { "dtype": "float32", "shape": [2, 2], "data": { "kind": "constant", "value": 2.0 } },
"B": { "dtype": "float32", "shape": [2, 2], "data": { "kind": "constant", "value": 3.0 } }
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 2],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [0.0, 0.0, 0.0, 0.0] }
}
}
},
{
"name": "ort_float32_empty_k_dimension_outputs_zero",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.DoubleTypeEmptyKDim",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": { "dtype": "float32", "shape": [2, 0], "data": { "kind": "values", "values": [] } },
"B": { "dtype": "float32", "shape": [0, 3], "data": { "kind": "values", "values": [] } }
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 3],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [0.0, 0.0, 0.0, 0.0, 0.0, 0.0] }
}
}
},
{
"name": "ort_float32_transpose_a_scaled",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.DoubleTypeScale",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"attrs": { "alpha": 0.5, "transA": 1 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 3],
"data": { "kind": "values", "values": [1.0, 2.0, 3.0, 4.0, 5.0, 6.0] }
},
"B": {
"dtype": "float32",
"shape": [2, 3],
"data": { "kind": "values", "values": [7.0, 8.0, 9.0, 10.0, 11.0, 12.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [3, 3],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [23.5, 26.0, 28.5, 32.0, 35.5, 39.0, 40.5, 45.0, 49.5] }
}
}
},
{
"name": "ort_float32_transpose_b",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeTransposeB",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"attrs": { "transB": 1 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 3],
"data": { "kind": "values", "values": [1.0, 2.0, 3.0, 4.0, 5.0, 6.0] }
},
"B": {
"dtype": "float32",
"shape": [4, 3],
"data": { "kind": "values", "values": [7.0, 8.0, 9.0, 10.0, 11.0, 12.0, 13.0, 14.0, 15.0, 16.0, 17.0, 18.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 4],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [50.0, 68.0, 86.0, 104.0, 122.0, 167.0, 212.0, 257.0] }
}
}
},
{
"name": "ort_float32_transpose_ab_scaled",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeScale",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"attrs": { "alpha": 4, "transA": 1, "transB": 1 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [3, 2],
"data": { "kind": "values", "values": [1.0, 4.0, 2.0, 5.0, 3.0, 6.0] }
},
"B": {
"dtype": "float32",
"shape": [4, 3],
"data": { "kind": "values", "values": [7.0, 8.0, 9.0, 10.0, 11.0, 12.0, 13.0, 14.0, 15.0, 16.0, 17.0, 18.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 4],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [200.0, 272.0, 344.0, 416.0, 488.0, 668.0, 848.0, 1028.0] }
}
}
},
{
"name": "ort_float32_scaled_no_transpose",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeScale",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 3],
"data": { "kind": "values", "values": [1.0, 2.0, 3.0, 4.0, 5.0, 6.0] }
},
"B": {
"dtype": "float32",
"shape": [3, 2],
"data": { "kind": "values", "values": [7.0, 8.0, 9.0, 10.0, 11.0, 12.0] }
}
},
"outputs": {
"Y": {
"dtype": "float32",
"shape": [2, 2],
"tolerance": 0.000001,
"data": { "kind": "values", "values": [29.0, 32.0, 69.5, 77.0] }
}
}
},
{
"name": "ort_float32_empty_input_m_zero",
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.DoubleTypeEmptyInput",
"notes": "Derived from the com.microsoft.FusedMatMul corpus. Upstream registers one kernel for both operators and TransposeMatMul is the strict subset that omits transBatchA/transBatchB, so an expectation taken with those flags at zero describes this operator exactly."
},
"inputs": {
"A": { "dtype": "float32", "shape": [0, 3], "data": { "kind": "values", "values": [] } },
"B": {
"dtype": "float32",
"shape": [3, 4],
"data": { "kind": "values", "values": [1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0] }
}
},
"outputs": {
"Y": { "dtype": "float32", "shape": [0, 4], "tolerance": 0.000001, "data": { "kind": "values", "values": [] } }
}
},
{
"name": "aligned_plain_64x32x64",
"inputs": {
"A": {
"dtype": "float32",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [64, 64], "tolerance": 0.0001 } }
},
{
"name": "register_blocked_plain_512x64x512_alpha_scaled",
"inputs": {
"A": {
"dtype": "float32",
"shape": [512, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [64, 512],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [512, 512], "tolerance": 0.0001 } },
"attrs": { "alpha": 0.5 },
"provenance": {
"notes": "Rank-2 M=N=512 and K=64 produce 64 aligned 64x64 workgroup tiles, exercising register-blocked vec4 staging and 4x4 per-thread accumulation. alpha=0.5 verifies scaling in the output epilogue."
}
},
{
"name": "aligned_transB_alpha_64x32",
"attrs": { "transB": 1, "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [64, 64], "tolerance": 0.0001 } }
},
{
"name": "aligned_batched_plain_2x64x32x64",
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [2, 32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 64, 64], "tolerance": 0.0001 } }
},
{
"name": "aligned_batched_transB_alpha_2x64x32",
"attrs": { "transB": 1, "alpha": 0.25 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [2, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 64, 64], "tolerance": 0.0001 } }
},
{
"name": "aligned_mtail_50x32x128",
"inputs": {
"A": {
"dtype": "float32",
"shape": [50, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.015, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [32, 128],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.011, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [50, 128], "tolerance": 0.0001 } }
},
{
"name": "aligned_transA_64x32",
"attrs": { "transA": 1 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [64, 64], "tolerance": 0.0001 } }
},
{
"name": "aligned_transA_transB_alpha_64x32",
"attrs": { "transA": 1, "transB": 1, "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [64, 64], "tolerance": 0.0001 } }
},
{
"name": "f32_subgroup_matrix_subnormal_dot_products_gpu_gap",
"skipGpu": {
"category": "permanent",
"reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; the ~3e-39 subnormal dot products collapse to zero (subgroup-matrix path)."
},
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeNoTranspose",
"notes": "M=32, K=32, N=64 with finite subnormal dot products that must not flush to zero."
},
"inputs": {
"A": { "dtype": "float32", "shape": [32, 32], "data": { "kind": "constant", "value": 1e-20 } },
"B": { "dtype": "float32", "shape": [32, 64], "data": { "kind": "constant", "value": 1e-20 } }
},
"outputs": { "Y": { "dtype": "float32", "shape": [32, 64], "tolerance": 1e-43 } }
},
{
"name": "f32_subgroup_matrix_scaled_subnormal_dot_products_gpu_gap",
"skipGpu": {
"category": "permanent",
"reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; the ~3e-39 subnormal dot products collapse to zero before alpha scaling (subgroup-matrix path)."
},
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeScale",
"notes": "Alpha scaling is applied after accumulation, so finite subnormal products remain valid nonzero outputs."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": { "dtype": "float32", "shape": [32, 32], "data": { "kind": "constant", "value": 1e-20 } },
"B": { "dtype": "float32", "shape": [32, 64], "data": { "kind": "constant", "value": 1e-20 } }
},
"outputs": { "Y": { "dtype": "float32", "shape": [32, 64], "tolerance": 1e-43 } }
},
{
"name": "f32_subgroup_matrix_transB_subnormal_dot_products_gpu_gap",
"skipGpu": {
"category": "permanent",
"reason": "Portable WGSL floating-point semantics do not guarantee preservation of the subnormal values required by this fixture. Backend evidence: WebGPU/Metal flushes subnormals to zero in f32; the ~3e-39 subnormal dot products collapse to zero (subgroup-matrix transB path)."
},
"provenance": {
"source": "onnxruntime/test/contrib_ops/fused_matmul_op_test.cc",
"test": "FusedMatMulOpTest.FloatTypeTransposeB",
"notes": "The transposed-B subgroup-matrix path has the same finite subnormal accumulation requirement."
},
"attrs": { "transB": 1 },
"inputs": {
"A": { "dtype": "float32", "shape": [32, 32], "data": { "kind": "constant", "value": 1e-20 } },
"B": { "dtype": "float32", "shape": [64, 32], "data": { "kind": "constant", "value": 1e-20 } }
},
"outputs": { "Y": { "dtype": "float32", "shape": [32, 64], "tolerance": 1e-43 } }
},
{
"name": "aligned_f16_plain_64x32x64",
"inputs": {
"A": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.002 } }
},
{
"name": "aligned_f16_transB_alpha_64x32",
"attrs": { "transB": 1, "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.02 } }
},
{
"name": "f16_unaligned_3x5x7",
"inputs": {
"A": {
"dtype": "float16",
"shape": [3, 5],
"data": { "kind": "fillFloat32", "sinStep": 0.015, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [5, 7],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.011, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [3, 7], "tolerance": 0.003 } }
},
{
"name": "aligned_f16_transA_64x32",
"attrs": { "transA": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.001 } }
},
{
"name": "aligned_f16_batched_plain_2x64x32x64",
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [2, 32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 64, 64], "tolerance": 0.002 } }
},
{
"name": "f16_rank3_by_broadcast_rank3",
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 2, 3],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [1, 3, 4],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 2, 4], "tolerance": 0.002 } }
},
{
"name": "aligned_f16_transA_transB_alpha_64x32",
"attrs": { "transA": 1, "transB": 1, "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.002 } }
},
{
"name": "subgroup_matrix_m_tail_57_partial_block_f16",
"inputs": {
"A": {
"dtype": "float16",
"shape": [57, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.011, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [57, 64], "tolerance": 0.001 } }
},
{
"name": "subgroup_matrix_m_tail_33_alpha_scaled_f32",
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [33, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.015, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.019, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [33, 64], "tolerance": 0.0002 } }
},
{
"name": "empty_n_dimension_zero_width_output",
"inputs": {
"A": {
"dtype": "float32",
"shape": [3, 4],
"data": { "kind": "values", "values": [1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0, 11.0, 12.0] }
},
"B": { "dtype": "float32", "shape": [4, 0], "data": { "kind": "values", "values": [] } }
},
"outputs": {
"Y": { "dtype": "float32", "shape": [3, 0], "tolerance": 0.000001, "data": { "kind": "values", "values": [] } }
}
},
{
"name": "transA_transB_subgroup_matrix_m_tail_50_f16",
"attrs": { "transA": 1, "transB": 1, "alpha": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [32, 50],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [50, 64], "tolerance": 0.002 } }
},
{
"name": "f32_decode_gemv_m1_k65_n68_vec4_compact",
"provenance": {
"notes": "A compact M=1 float32 matrix product with odd K and N=68 checks the reduction and final output-column tail."
},
"attrs": { "alpha": 1 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [1, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [65, 68],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [1, 68], "tolerance": 0.0002 } }
},
{
"name": "f16_decode_gemv_m1_k65_n68_vec4_compact",
"provenance": {
"notes": "A float16 M=1 matrix product with odd K and N=68 checks both reduction and output-column tails. The dot product accumulates in float32 and narrows only at the store."
},
"attrs": { "alpha": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [65, 68],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 68], "tolerance": 0.004, "relTolerance": 0.004 } }
},
{
"name": "f16_decode_gemv_m1_k64_n128_alpha_half",
"provenance": {
"notes": "A float16 M=1 matrix product with even K and non-unit alpha checks complete reduction. The dot product accumulates in float32 and narrows only at the store."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [64, 128],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 128], "tolerance": 0.0002, "relTolerance": 0.004 } }
},
{
"name": "f32_rank4_by_rank2_shared_weight_compact",
"provenance": {
"notes": "A rank-2 weight shared across a 2x3 batch multiplies M=5, K=7, N=9 (alpha=0.5); the odd dimensions check batch-offset indexing and non-power-of-two tails."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 3, 5, 7],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [7, 9],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 3, 5, 9], "tolerance": 0.0002 } }
},
{
"name": "subgroup_matrix_kn_tail_f16_compact",
"provenance": {
"notes": "M=33 and K=34 are one and two past a 32-element boundary and N=66 is two past a 64-element boundary (float16, alpha=0.5), checking partial tiles in all three dimensions."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [33, 34],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.1 }
},
"B": {
"dtype": "float16",
"shape": [34, 66],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.1 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [33, 66], "tolerance": 0.0004 } }
},
{
"name": "subgroup_matrix_broadcast_rank4x3_f16_compact",
"provenance": {
"notes": "A rank-4 [1,2,33,32] by rank-3 [2,32,64] float16 broadcast multiplies M=33, K=32, N=64 across a batch of 2, checking batch-dimension broadcasting at a compact scale."
},
"attrs": { "alpha": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 2, 33, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.1 }
},
"B": {
"dtype": "float16",
"shape": [2, 32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.1 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 2, 33, 64], "tolerance": 0.0005 } }
},
{
"name": "broadcast_rank4_tiled_reg_f16_compact",
"provenance": {
"notes": "Compact rank-4 by rank-3 float16 broadcast with odd M/K/N checks output and reduction tails."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 2, 65, 33],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.1 }
},
"B": {
"dtype": "float16",
"shape": [2, 33, 67],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.1 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 2, 65, 67], "tolerance": 0.0004 } }
},
{
"name": "broadcast_rank4_tiled_reg_shared_f32_compact",
"provenance": {
"notes": "Float32 shared rank-2 weight counterpart for the register-blocked rank-4 path. Odd M/K/N cover all output and reduction tails."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [1, 2, 65, 33],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.1 }
},
"B": {
"dtype": "float32",
"shape": [33, 67],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.1 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [1, 2, 65, 67], "tolerance": 0.0003 } }
},
{
"name": "rank5_three_batch_dims",
"attrs": { "alpha": 1, "transA": 0, "transB": 0 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 1, 2, 2, 3],
"data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.5 }
},
"B": {
"dtype": "float32",
"shape": [1, 3, 1, 3, 4],
"data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31, "scale": 0.25 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 3, 2, 2, 4], "tolerance": 0.000001 } }
},
{
"name": "subgroup_matrix_kn_tail_f16_offset_alpha_scale",
"provenance": {
"notes": "Offset float16 operands keep each output near `alpha * K * aOffset * bOffset` (about 3.4), making the K=34 reduction tail, N=66 column tail, and `alpha = 0.5` epilogue observable on the subgroup-matrix route."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [33, 34],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.1, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [34, 66],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.1, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [33, 66], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "subgroup_matrix_broadcast_rank4x3_f16_offset_scale",
"provenance": {
"notes": "Offset operands keep outputs near `K * aOffset * bOffset`, making the per-batch B slice and K=32 contraction observable in a rank-4 by rank-3 broadcast."
},
"attrs": { "alpha": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 2, 33, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.1, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [2, 32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.1, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 2, 33, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "subgroup_matrix_a_batch_broadcast_rank4x3_f16",
"provenance": {
"notes": "A has batch extent 1 while B has extent 2, so one A slice feeds both output batches. Offset operands keep the expected magnitude nonzero, exposing a swapped or nonzero A batch stride."
},
"attrs": { "alpha": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 1, 33, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.019, "scale": 0.1, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [2, 32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.007, "cosStep": 0.031, "scale": 0.1, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 2, 33, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "broadcast_rank4_tiled_reg_f16_offset_alpha_scale",
"provenance": {
"notes": "Offset operands keep outputs near `alpha * K * aOffset * bOffset` with K=33, making the one-element reduction tail and `alpha` multiplier observable on the register-blocked rank-4 route."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 2, 65, 33],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.1, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [2, 33, 67],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.1, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 2, 65, 67], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "aligned_f16_transA_transB_alpha_offset_scale",
"provenance": {
"notes": "Offset operands keep the doubly transposed output proportional to `alpha * K`, making the K=32 contraction and `alpha = 0.5` epilogue observable on subgroup-matrix and portable tiled routes."
},
"attrs": { "transA": 1, "transB": 1, "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "f16_unaligned_3x5x7_offset_scale",
"provenance": {
"notes": "Offset operands in a 3x5 by 5x7 multiply keep outputs proportional to K, exposing dropped reduction elements or doubled tails on the unaligned scalar and tiled routes."
},
"inputs": {
"A": {
"dtype": "float16",
"shape": [3, 5],
"data": { "kind": "fillFloat32", "sinStep": 0.015, "cosStep": 0.021, "scale": 0.2, "offset": 0.6 }
},
"B": {
"dtype": "float16",
"shape": [5, 7],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.011, "scale": 0.2, "offset": 0.5 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [3, 7], "tolerance": 0.01, "relTolerance": 0.005 } }
},
{
"name": "aligned_f16_plain_64x32x64_offset_scale",
"provenance": {
"notes": "Offset operands keep each fully aligned M=64, K=32, N=64 output proportional to K, making the subgroup-matrix reduction count and scratch drain observable."
},
"inputs": {
"A": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "aligned_f16_batched_plain_2x64x32x64_offset_scale",
"provenance": {
"notes": "Two batches carry distinct offset operands, keeping outputs proportional to K and making both the batch stride and aligned subgroup-matrix reduction count observable."
},
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [2, 32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 64, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "subgroup_matrix_m_tail_57_partial_block_f16_offset_scale",
"provenance": {
"notes": "M=57 leaves 25 rows after one full 32-row tile. Offset operands require tail rows to match the full rows' expected magnitude, exposing a short reduction or stale scratch value."
},
"inputs": {
"A": {
"dtype": "float16",
"shape": [57, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.023, "scale": 0.2, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.011, "scale": 0.2, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [57, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "aligned_f16_transA_64x32_offset_scale",
"provenance": {
"notes": "With only A transposed, offset operands keep each output proportional to K and expose both a transposed-A stride error and an incorrect reduction count."
},
"attrs": { "transA": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [64, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "transA_transB_subgroup_matrix_m_tail_50_f16_offset_scale",
"provenance": {
"notes": "Both operands are transposed and M=50 leaves an 18-row tail. Offset operands make the guarded tail rows' magnitude and placement independently observable."
},
"attrs": { "transA": 1, "transB": 1, "alpha": 1 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [32, 50],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2, "offset": 0.5 }
},
"B": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2, "offset": 0.4 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [50, 64], "tolerance": 0.03, "relTolerance": 0.01 } }
},
{
"name": "f16_rank3_by_broadcast_rank3_offset_scale",
"provenance": {
"notes": "f16_rank3_by_broadcast_rank3 cancels to 0.076 under a 0.02 absolute tolerance (26% blind). Offsetting both operands makes each element ~K * aOffset * bOffset over K=3, so the shared single-batch B - read by both output batches - is pinned for value as well as for broadcast addressing."
},
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 2, 3],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.027, "scale": 0.2, "offset": 1.0 }
},
"B": {
"dtype": "float16",
"shape": [1, 3, 4],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.007, "scale": 0.2, "offset": 0.8 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 2, 4], "tolerance": 0.01, "relTolerance": 0.005 } }
},
{
"name": "rank6_four_batch_dims",
"attrs": { "alpha": 1, "transA": 0, "transB": 0 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 2, 1, 2, 2, 3],
"data": { "kind": "fillFloat32", "sinStep": 0.13, "cosStep": 0.29, "scale": 0.5 }
},
"B": {
"dtype": "float32",
"shape": [1, 1, 3, 1, 3, 4],
"data": { "kind": "fillFloat32", "sinStep": 0.17, "cosStep": 0.31, "scale": 0.25 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 2, 3, 2, 2, 4], "tolerance": 0.000001 } }
},
{
"name": "subgroup_matrix_band_m8_f16",
"attrs": { "alpha": 2 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [8, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021 }
},
"B": {
"dtype": "float16",
"shape": [64, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.029 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [8, 64], "tolerance": 0.005 } }
},
{
"name": "subgroup_matrix_splitk_m_tail_f16",
"attrs": { "alpha": 2 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [10, 2048],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021 }
},
"B": {
"dtype": "float16",
"shape": [2048, 256],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.031 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [10, 256], "tolerance": 0.005 } }
},
{
"name": "subgroup_matrix_splitk_alpha_scaled_f32",
"attrs": { "alpha": 1.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [16, 1024],
"data": { "kind": "fillFloat32", "sinStep": 0.03, "cosStep": 0.07, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [1024, 128],
"data": { "kind": "fillFloat32", "sinStep": 0.041, "cosStep": 0.089, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [16, 128], "tolerance": 0.0002 } }
},
{
"name": "band_vec4_alpha_scaled_m8_k256_n512",
"provenance": {
"notes": "Eight rows, 256 reduction elements, and 512 output columns with alpha=0.5 verify that the non-unit scale factor is applied correctly to the matrix product."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [8, 256],
"data": { "kind": "fillFloat32", "sinStep": 0.07, "cosStep": 0.13, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [256, 512],
"data": { "kind": "fillFloat32", "sinStep": 0.05, "cosStep": 0.19, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [8, 512], "tolerance": 0.0001 } }
},
{
"name": "subgroup_matrix_batched_transB_small_m_f16",
"attrs": { "transB": 1, "alpha": 0.25 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 4, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.011, "cosStep": 0.023, "scale": 0.2 }
},
"B": {
"dtype": "float16",
"shape": [2, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.009, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 4, 64], "tolerance": 0.01 } }
},
{
"name": "subgroup_matrix_transA_small_m_f16",
"attrs": { "transA": 1, "alpha": 3 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [64, 8],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021 }
},
"B": {
"dtype": "float16",
"shape": [64, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.029 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [8, 64], "tolerance": 0.005 } }
},
{
"name": "f32_decode_gemv_m1_k65_n68_alpha_half",
"provenance": {
"notes": "A compact M=1 GEMV with alpha=0.5 exercises a non-unit multiplier folded into the store as a baked constant. The expected output verifies that the multiplier is applied exactly once."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [1, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [65, 68],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [1, 68], "tolerance": 0.0002 } }
},
{
"name": "f16_rank4_by_rank2_shared_weight_m33_k34_n66",
"provenance": {
"notes": "A [2, 3, 33, 34] float16 A shares one [34, 66] B across both batch axes, with alpha 0.5; none of M=33, K=34 or N=66 is a multiple of 32."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 3, 33, 34],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021 }
},
"B": {
"dtype": "float16",
"shape": [34, 66],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 3, 33, 66], "tolerance": 0.001 } }
},
{
"name": "f16_rank4_by_rank2_shared_weight_m8_k32_n64",
"provenance": {
"notes": "A [2, 3, 8, 32] float16 A shares one [32, 64] B across both batch axes, with alpha 0.5, at an 8-row M."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 3, 8, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 3, 8, 64], "tolerance": 0.001 } }
},
{
"name": "f16_rank4_by_rank2_shared_weight_m1_k32_n64",
"provenance": {
"notes": "A [2, 3, 1, 32] float16 A shares one [32, 64] B across both batch axes, with alpha 0.5, at a single-row M."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 3, 1, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.021 }
},
"B": {
"dtype": "float16",
"shape": [32, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.009, "cosStep": 0.029 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 3, 1, 64], "tolerance": 0.001 } }
},
{
"name": "f32_band_preferred_m4_k2048_n4096",
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [4, 2048],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.1, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [2048, 4096],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.041, "scale": 0.1, "offset": 0.03 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [4, 4096], "tolerance": 0.0001, "relTolerance": 0.00001 } },
"provenance": {
"notes": "M=4, K=2048, N=4096 float32 operands with alpha=0.5 and different per-operand offsets (0.02 vs 0.03) avoid cancellation in the matrix product."
}
},
{
"name": "f32_band_preferred_m16_k2560_n4096",
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [16, 2560],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.1, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [2560, 4096],
"data": { "kind": "fillFloat32", "sinStep": 0.017, "cosStep": 0.041, "scale": 0.1, "offset": 0.03 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [16, 4096], "tolerance": 0.0001, "relTolerance": 0.00001 } },
"provenance": {
"notes": "M=16, K=2560, N=4096 float32 operands with alpha=0.5 and different per-operand offsets (0.02 vs 0.03) avoid cancellation in the matrix product."
}
},
{
"name": "broadcast-transb-tails-float16",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 1, 129, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [3, 129, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 3, 129, 129], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-rank3x2-float16",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [4, 256, 128],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [512, 128],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [4, 256, 512], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-rank5x3-float16",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [2, 1, 2, 128, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [2, 128, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [2, 1, 2, 128, 128], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-low_tiles-float16",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [1, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [1, 64, 64], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-large-float32",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 8, 512, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [8, 512, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 8, 512, 512], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-tails-float32",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 1, 129, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [3, 129, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 3, 129, 129], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-rank3x2-float32",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [4, 256, 128],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [512, 128],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [4, 256, 512], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-rank5x3-float32",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [2, 1, 2, 128, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [2, 128, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [2, 1, 2, 128, 128], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-transb-low_tiles-float32",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [1, 64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [64, 32],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [1, 64, 64], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b3-m128-k64-n256-float16-a0.125",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [3, 128, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [256, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [3, 128, 256], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b4-m128-k64-n256-float16-a-0.375",
"attrs": { "transB": 1, "alpha": -0.375 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [4, 128, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [256, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [4, 128, 256], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b8-m128-k64-n256-float16-a0.5",
"attrs": { "transB": 1, "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [8, 128, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [256, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [8, 128, 256], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b8-m129-k65-n257-float16-a0.125",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [8, 129, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [257, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [8, 129, 257], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b8-m128-k96-n128-float16-a0",
"attrs": { "transB": 1, "alpha": 0 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [8, 128, 96],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [128, 96],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [8, 128, 128], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b4-m256-k127-n512-float16-a-0.5",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [4, 256, 127],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [512, 127],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [4, 256, 512], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b3-m128-k64-n256-float32-a0.125",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [3, 128, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [256, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [3, 128, 256], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b4-m128-k64-n256-float32-a-0.375",
"attrs": { "transB": 1, "alpha": -0.375 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [4, 128, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [256, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [4, 128, 256], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b8-m128-k64-n256-float32-a0.5",
"attrs": { "transB": 1, "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [8, 128, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [256, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [8, 128, 256], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b8-m129-k65-n257-float32-a0.125",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [8, 129, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [257, 65],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [8, 129, 257], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b8-m128-k96-n128-float32-a0",
"attrs": { "transB": 1, "alpha": 0 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [8, 128, 96],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [128, 96],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [8, 128, 128], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-grid-b4-m256-k127-n512-float32-a-0.5",
"attrs": { "transB": 1, "alpha": -0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [4, 256, 127],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [512, 127],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [4, 256, 512], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-selected-mn-tail-float16",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [8, 129, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [257, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [8, 129, 257], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-selected-mn-tail-float32",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [8, 129, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [257, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [8, 129, 257], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-selected-broadcast-tail-float16",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float16",
"shape": [4, 1, 129, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float16",
"shape": [3, 257, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float16", "shape": [4, 3, 129, 257], "tolerance": 0.001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "broadcast-selected-broadcast-tail-float32",
"attrs": { "transB": 1, "alpha": 0.125 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [4, 1, 129, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.013, "cosStep": 0.031, "scale": 0.075, "offset": 0.02 }
},
"B": {
"dtype": "float32",
"shape": [3, 257, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.019, "cosStep": 0.017, "scale": 0.075, "offset": -0.02 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [4, 3, 129, 257], "tolerance": 0.00001, "relTolerance": 0 } },
"provenance": { "notes": "Broadcast transposed-B geometry, alpha, padding and output-grid boundary." }
},
{
"name": "plain_rank2_tiled_reg_row_tail_m100_k64_n2048",
"provenance": {
"notes": "M=100 leaves 36 rows past a 64-row boundary (K=64, N=2048), with alpha=0.5 scaling the output; the extra rows must be included correctly in the result."
},
"attrs": { "alpha": 0.5 },
"inputs": {
"A": {
"dtype": "float32",
"shape": [100, 64],
"data": { "kind": "fillFloat32", "sinStep": 0.037, "cosStep": 0.061, "scale": 0.2 }
},
"B": {
"dtype": "float32",
"shape": [64, 2048],
"data": { "kind": "fillFloat32", "sinStep": 0.043, "cosStep": 0.079, "scale": 0.2 }
}
},
"outputs": { "Y": { "dtype": "float32", "shape": [100, 2048], "tolerance": 0.0001 } }
}
]
}