From fb13f490a10e5fa6b31d082a69ba9f2adc1a7df8 Mon Sep 17 00:00:00 2001 From: wcy Date: Wed, 2 Sep 2026 01:43:58 +0900 Subject: [PATCH 1/6] Rebase P4 MAC vectorization pass onto upstream HEAD (#70) Ports p4_mac_vectorization.patch (issue #69) onto eaf8d1c, which added the Appliance Mode compiler (#70) and refactored CLI options into a shared `codegen_options` decorator used by both `compile_spatial_ir` and the new `compile_spatial_ir_appliance`. The patch's CLI hunk no longer applied cleanly (2/3 hunks rejected) because compiler.py's option list and function signature moved; the lowering/statements.py hunks applied without conflict. Resolved by hand-adding --disable-mac-vectorization to the shared codegen_options decorator and threading disable_mac_vectorization through generate_program() and both CLI entry points (compile_spatial_ir, compile_spatial_ir_appliance), matching how --disable-dsd is already threaded through both. Baseline (upstream HEAD, no P4): 375 passed, 12 skipped. This commit: 381 passed, 12 skipped (0 regressions, +6 P4 tests). --- spada/cli/appliance_compiler.py | 2 + spada/cli/compiler.py | 8 +- spada/lowering/spatial_ir_to_csl.py | 8 +- spada/syntax/csl/statements.py | 186 +++++++++++++++++++++ tests/spatial_ir/test_mac_vectorization.py | 130 ++++++++++++++ 5 files changed, 331 insertions(+), 3 deletions(-) create mode 100644 tests/spatial_ir/test_mac_vectorization.py diff --git a/spada/cli/appliance_compiler.py b/spada/cli/appliance_compiler.py index f35037e9..b43ecd37 100644 --- a/spada/cli/appliance_compiler.py +++ b/spada/cli/appliance_compiler.py @@ -92,6 +92,7 @@ def compile_spatial_ir_appliance(input_file: str, output_folder: str, param: lis generate_only: bool, simulator: bool, mgmt_namespace: Optional[str], resource_cpu: Optional[int], resource_mem: Optional[int], check_version: bool, disable_benchmarking: bool, disable_asynchronous: bool, disable_dsd: bool, + disable_mac_vectorization: bool, disable_map: bool, disable_task_fusion: bool, disable_task_recycling: bool, disable_copy_elision: bool, disable_close_elision: bool, disable_switching: bool): program = generate_program( @@ -103,6 +104,7 @@ def compile_spatial_ir_appliance(input_file: str, output_folder: str, param: lis disable_benchmarking=disable_benchmarking, disable_asynchronous=disable_asynchronous, disable_dsd=disable_dsd, + disable_mac_vectorization=disable_mac_vectorization, disable_map=disable_map, disable_task_fusion=disable_task_fusion, disable_task_recycling=disable_task_recycling, diff --git a/spada/cli/compiler.py b/spada/cli/compiler.py index 46785c46..723e53ab 100644 --- a/spada/cli/compiler.py +++ b/spada/cli/compiler.py @@ -62,6 +62,8 @@ def codegen_options(func): help='Disable benchmarking code generation (and memory overhead)'), click.option('--disable-asynchronous', is_flag=True, help='Disable asynchronous task code generation'), click.option('--disable-dsd', is_flag=True, help='Disable DSD operation detection and code generation'), + click.option('--disable-mac-vectorization', is_flag=True, + help='Disable vectorization of nested multiply-accumulate loops into @fmac* DSD operations'), click.option('--disable-map', is_flag=True, help='Disable @map operation detection and code generation'), click.option('--disable-task-fusion', is_flag=True, help='Disable task fusion optimization'), click.option('--disable-task-recycling', is_flag=True, help='Disable task ID recycling'), @@ -76,6 +78,7 @@ def codegen_options(func): def generate_program(input_file: str, output_folder: str, param: list[str] = (), offset_x: int = 0, offset_y: int = 0, disable_benchmarking: bool = False, disable_asynchronous: bool = False, disable_dsd: bool = False, + disable_mac_vectorization: bool = False, disable_map: bool = False, disable_task_fusion: bool = False, disable_task_recycling: bool = False, disable_copy_elision: bool = False, disable_close_elision: bool = False, disable_switching: bool = False) -> GeneratedProgram: @@ -138,6 +141,7 @@ def generate_program(input_file: str, output_folder: str, param: list[str] = (), disable_benchmarking=disable_benchmarking, disable_asynchronous=disable_asynchronous, disable_dsd=disable_dsd, + disable_mac_vectorization=disable_mac_vectorization, task_fusion=not disable_task_fusion, copy_elision=not disable_copy_elision, task_id_recycling=not disable_task_recycling, @@ -256,7 +260,8 @@ def cslc_arguments(program: GeneratedProgram, hardware_fabric: Optional[bool] = @click.option('--generate-only', '-g', is_flag=True, help='Only generate the output files without compiling them') def compile_spatial_ir(input_file: str, output_folder: str, param: list[str], offset_x: int, offset_y: int, generate_only: bool, disable_benchmarking: bool, - disable_asynchronous: bool, disable_dsd: bool, disable_map: bool, + disable_asynchronous: bool, disable_dsd: bool, disable_mac_vectorization: bool, + disable_map: bool, disable_task_fusion: bool, disable_task_recycling: bool, disable_copy_elision: bool, disable_close_elision: bool, disable_switching: bool): program = generate_program( @@ -268,6 +273,7 @@ def compile_spatial_ir(input_file: str, output_folder: str, param: list[str], of disable_benchmarking=disable_benchmarking, disable_asynchronous=disable_asynchronous, disable_dsd=disable_dsd, + disable_mac_vectorization=disable_mac_vectorization, disable_map=disable_map, disable_task_fusion=disable_task_fusion, disable_task_recycling=disable_task_recycling, diff --git a/spada/lowering/spatial_ir_to_csl.py b/spada/lowering/spatial_ir_to_csl.py index 4c15a246..cadb29ce 100644 --- a/spada/lowering/spatial_ir_to_csl.py +++ b/spada/lowering/spatial_ir_to_csl.py @@ -54,6 +54,7 @@ def lower_spatial_ir_to_csl(kernel: spir.Kernel, disable_benchmarking: bool = False, disable_asynchronous: bool = False, disable_dsd: bool = False, + disable_mac_vectorization: bool = False, task_fusion: bool = True, copy_elision: bool = True, prune_memory: bool = True, @@ -158,7 +159,7 @@ def lower_spatial_ir_to_csl(kernel: spir.Kernel, rect_code, color_map = generate_rectangle(kernel, rect, routing_instructions, scalar_arguments, use_memcpy_mode, stream_rects, channel_to_color, disable_benchmarking, disable_asynchronous, disable_dsd, task_fusion, - task_id_recycling) + task_id_recycling, disable_mac_vectorization) color_maps.append(color_map) csl_codes.append(CodeFile(csl_name, rect_code)) @@ -318,7 +319,8 @@ def generate_rectangle(kernel: spir.Kernel, disable_asynchronous: bool = False, disable_dsd: bool = False, task_fusion: bool = True, - task_id_recycling: bool = True) -> tuple[str, dict[str, int]]: + task_id_recycling: bool = True, + disable_mac_vectorization: bool = False) -> tuple[str, dict[str, int]]: # Code generation carets header = StringIO() current_code = StringIO() @@ -326,6 +328,8 @@ def generate_rectangle(kernel: spir.Kernel, if disable_dsd: dsd_ops.DISABLE_DSD = True + if disable_mac_vectorization: + cslstmt.DISABLE_MAC_VECTORIZATION = True header.write(""" param memcpy_params; diff --git a/spada/syntax/csl/statements.py b/spada/syntax/csl/statements.py index 31761b9a..d05039db 100644 --- a/spada/syntax/csl/statements.py +++ b/spada/syntax/csl/statements.py @@ -250,6 +250,187 @@ def emit_assignment(statement: spir.AssignmentStatement, dsds: UniqueDSDDict, dt return dsd_ops.DSD_ASSIGNMENT_MAPPING[dsd_op]() +# [P4] Targeted vectorization of the nested multiply-accumulate loop +# for (k, l) in [0:Kk, 0:Kl]: Z[k] = Z[k] + A[k*Ck + l*Cl + C0] * X[f(l)] +# into the handwritten CSL idiom (strided base DSD + per-l @increment_dsd_offset + @fmacs). +# Conservative: fires only on the exact f32 MAC shape with Z distinct from A and X; +# anything else falls through to scalar loops. @increment_dsd_offset is available from +# SDK 1.x (used by Cerebras' own csl-examples v1.4.0 cholesky benchmark). +DISABLE_MAC_VECTORIZATION = False +_VEC_COUNTER = [0] + + +def _affine_of(expr, varnames): + """expr 를 sum(coeff[var]*var) + const 로 분해. 실패 시 None. 계수·상수는 int.""" + e = expr.value if isinstance(expr, spir.Expression) else expr + if isinstance(e, spir.Identifier): + if e.name in varnames: + return {e.name: 1}, 0 + return None + if isinstance(e, (spir.ConstantLiteral, spir.Parameter)): + try: + v = e.eval() + except Exception: + return None + return ({}, int(v)) if isinstance(v, int) else None + if isinstance(e, spir.UnaryOperator) and getattr(e, 'op', None) == '-': + sub = _affine_of(e.operand, varnames) if hasattr(e, 'operand') else None + if sub is None: + return None + return {k: -c for k, c in sub[0].items()}, -sub[1] + if isinstance(e, spir.BinaryOperator): + left = _affine_of(e.left, varnames) + right = _affine_of(e.right, varnames) + if e.op == '+' and left and right: + coeffs = dict(left[0]) + for k, c in right[0].items(): + coeffs[k] = coeffs.get(k, 0) + c + return coeffs, left[1] + right[1] + if e.op == '-' and left and right: + coeffs = dict(left[0]) + for k, c in right[0].items(): + coeffs[k] = coeffs.get(k, 0) - c + return coeffs, left[1] - right[1] + if e.op == '*' and left and right: + if not left[0]: # 왼쪽이 순수 상수 + s = left[1] + return {k: c * s for k, c in right[0].items()}, right[1] * s + if not right[0]: # 오른쪽이 순수 상수 + s = right[1] + return {k: c * s for k, c in left[0].items()}, left[1] * s + return None + return None + + +def _ids_in(expr): + e = expr.value if isinstance(expr, spir.Expression) else expr + return {n.name for n in e.walk() if isinstance(n, spir.Identifier)} + + +def _single_index(node): + """ArraySlice 가 1차원 단일 인덱스 접근이면 그 인덱스 Expression, 아니면 None.""" + if not isinstance(node, spir.ArraySlice) or len(node.indices) != 1: + return None + idx = node.indices[0] + if isinstance(idx, spir.RangeExpression): + return None + return idx + + +def _try_emit_vectorized_mac(statement: spir.ForStatement, dtypes: dict[spir.Identifier, spir.IRType], + header_code: StringIO): + if len(statement.variables) != 2 or len(statement.range_expression) != 2: + return None + if len(statement.body) != 1 or not isinstance(statement.body[0], spir.AssignmentStatement): + return None + k_var = statement.variables[0].identifier.name + l_var = statement.variables[1].identifier.name + try: + rk, rl = statement.range_expression + k0 = rk.start.eval() if rk.start is not None else 0 + kk = rk.stop.eval() + ks = rk.step.eval() if rk.step is not None else 1 + l0 = rl.start.eval() if rl.start is not None else 0 + ll = rl.stop.eval() + ls = rl.step.eval() if rl.step is not None else 1 + except Exception: + return None + if not all(isinstance(v, int) for v in (k0, kk, ks, l0, ll, ls)) or (k0, ks, ls) != (0, 1, 1): + return None + + assign = statement.body[0] + dst = assign.destination + dst_idx = _single_index(dst) + if dst_idx is None: + return None + dst_aff = _affine_of(dst_idx, {k_var}) + if dst_aff != ({k_var: 1}, 0): + return None + + src = assign.source.value + if isinstance(src, spir.MultiplyAccumulateOperator): + acc, mul_b, mul_c = src.a.value, src.b.value, src.c.value + elif isinstance(src, spir.BinaryOperator) and src.op == '+': + acc = src.left.value + mul = src.right.value + if not (isinstance(mul, spir.BinaryOperator) and mul.op == '*'): + acc, mul = src.right.value, src.left.value + if not (isinstance(mul, spir.BinaryOperator) and mul.op == '*'): + return None + mul_b, mul_c = mul.left.value, mul.right.value + else: + return None + + # 누산 항은 목적지와 동일한 Z[k] + if not (isinstance(acc, spir.ArraySlice) and acc.array.name == dst.array.name): + return None + acc_idx = _single_index(acc) + if acc_idx is None or _affine_of(acc_idx, {k_var}) != ({k_var: 1}, 0): + return None + + # 곱 항: 하나는 k·l 아핀 접근의 배열 A, 하나는 l 만의 스칼라 접근 X + def classify(node): + if not isinstance(node, spir.ArraySlice): + return None + idx = _single_index(node) + if idx is None: + return None + aff = _affine_of(idx, {k_var, l_var}) + if aff is None: + return None + coeffs, const = aff + if coeffs.get(k_var) and coeffs[k_var] > 0: + return ('mat', node, coeffs.get(k_var), coeffs.get(l_var, 0), const) + if k_var not in coeffs: + return ('vec', node, idx) + return None + + cb, cc = classify(mul_b), classify(mul_c) + if cb and cb[0] == 'mat' and cc and cc[0] == 'vec': + mat, vec = cb, cc + elif cc and cc[0] == 'mat' and cb and cb[0] == 'vec': + mat, vec = cc, cb + else: + return None + _, mat_node, ck, cl, c0 = mat + _, vec_node, vec_idx = vec + if _ids_in(vec_idx) - {l_var} != set(): + return None + + # Aliasing guard: the DSD op reorders reads relative to the sequential scalar loop, + # so reading the written array through A or X must not be vectorized. + if mat_node.array.name == dst.array.name or vec_node.array.name == dst.array.name: + return None + + # f32 전용 (@fmacs) + def base_f32(name_node): + t = dtypes.get(name_node.array) + base = getattr(t, 'base_type', None) or getattr(t, 'element_type', None) + return base == spir.ScalarType.f32 + if not (base_f32(dst) and base_f32(mat_node) and base_f32(vec_node)): + return None + + uid = _VEC_COUNTER[0] + _VEC_COUNTER[0] += 1 + z_name = name_to_csl(dst.array) + a_name = name_to_csl(mat_node.array) + x_l = expr_to_csl(vec_idx) + dst_dsd = f'__vecmac_dst_{uid}' + src_dsd = f'__vecmac_src_{uid}' + base_expr = f'__index * {ck}' + (f' + {c0}' if c0 else '') + header_code.write( + f'const {dst_dsd} = @get_dsd(mem1d_dsd, .{{ .tensor_access = |__index|{{{kk}}} -> {z_name}[__index] }});\n' + f'const {src_dsd} = @get_dsd(mem1d_dsd, .{{ .tensor_access = |__index|{{{kk}}} -> {a_name}[{base_expr}] }});\n') + off = f'{name_to_csl(statement.variables[1].identifier)}' + (f' * {cl}' if cl != 1 else '') + lname = name_to_csl(statement.variables[1].identifier) + return ( + f'// [P4] vectorized MAC: {z_name}[k] += {a_name}[k*{ck}+{lname}*{cl}+{c0}] * {name_to_csl(vec_node.array)}[{x_l}]\n' + f'for (@range(i16, {l0}, {ll}, 1)) |{lname}| {{\n' + f' const __vecmac_a_{uid} = @increment_dsd_offset({src_dsd}, {off}, f32);\n' + f' @fmacs({dst_dsd}, {dst_dsd}, __vecmac_a_{uid}, {name_to_csl(vec_node.array)}[{x_l}]);\n' + f'}}\n') + + def emit_for(statement: spir.ForStatement, dsds: UniqueDSDDict, dtypes: dict[spir.Identifier, spir.IRType], header_code: StringIO) -> str: """ @@ -261,6 +442,11 @@ def emit_for(statement: spir.ForStatement, dsds: UniqueDSDDict, dtypes: dict[spi :param header_code: The header code to include. :return: The generated CSL for loop statement. """ + if not dsd_ops.DISABLE_DSD and not DISABLE_MAC_VECTORIZATION: + vectorized = _try_emit_vectorized_mac(statement, dtypes, header_code) + if vectorized is not None: + return vectorized + ranges = statement.range_expression vars_ = statement.variables diff --git a/tests/spatial_ir/test_mac_vectorization.py b/tests/spatial_ir/test_mac_vectorization.py new file mode 100644 index 00000000..ffd96266 --- /dev/null +++ b/tests/spatial_ir/test_mac_vectorization.py @@ -0,0 +1,130 @@ +"""Tests for the targeted vectorization of nested multiply-accumulate loops. + +The lowering should turn + + for (k, l) in [0:K, 0:K]: + z[k] = z[k] + A[k*K + l] * x[l] + +into a strided base DSD plus a per-column ``@increment_dsd_offset`` and ``@fmacs``, +and must fall back to scalar loops for every shape it cannot prove safe +(aliasing, non-affine indices, non-f32 dtypes, or the explicit disable flag). +""" +import pytest +from spada.lowering.spatial_ir_to_csl import lower_spatial_ir_to_csl +from spada.syntax.csl import statements as cslstmt +from spada.syntax.spatial_ir import parser, passes + + +def _kernel(body: str, decls: str = None) -> str: + decls = decls if decls is not None else ''' + f32[K*K] A_flat + f32[K] x + f32[K] z + ''' + return f''' + kernel @t( + stream[2, 2] readonly inp_A, + stream[2, 2] readonly inp_x, + stream[2, 2] writeonly out + ) {{ + place i16 i, i16 j in [0:2, 0:2] {{ + {decls} + }} + phase {{ + compute i16 i, i16 j in [0:2, 0:2] {{ + await receive(A_flat, inp_A[i, j]) + await receive(x, inp_x[i, j]) + {body} + await send(z, out[i, j]) + }} + }} + }} + ''' + + +def _lower(code: str, **lower_kwargs): + kernel = parser.parse_string(code, 'test.sptl') + kernel = passes.concretize_parameters(kernel, K=4) + kernel = passes.constexpr_propagation(kernel) + return lower_spatial_ir_to_csl(kernel, **lower_kwargs) + + +def _all_code(csl_files) -> str: + return '\n'.join(f.code for f in csl_files) + + +MAC_BODY = ''' + for i16 k in [0:K] { + z[k] = 0.0 + } + for i16 k, i16 l in [0:K, 0:K] { + z[k] = z[k] + A_flat[k*K + l] * x[l] + } +''' + + +def test_mac_loop_is_vectorized(): + code = _all_code(_lower(_kernel(MAC_BODY))) + assert '@fmacs(' in code + assert '@increment_dsd_offset(' in code + + +def test_scalar_fallback_when_disabled(): + try: + code = _all_code(_lower(_kernel(MAC_BODY), disable_mac_vectorization=True)) + assert '@increment_dsd_offset(' not in code + finally: + cslstmt.DISABLE_MAC_VECTORIZATION = False # module flag is sticky, reset for other tests + + +def test_no_vectorization_when_accumulator_aliases_matrix(): + # z appears as the matrix operand: DSD reordering would change semantics. + body = ''' + for i16 k in [0:K] { + z[k] = 1.0 + } + for i16 k, i16 l in [0:K, 0:K] { + z[k] = z[k] + z[k*1 + l*0] * x[l] + } + ''' + code = _all_code(_lower(_kernel(body))) + assert '@increment_dsd_offset(' not in code + + +def test_no_vectorization_when_accumulator_aliases_vector(): + body = ''' + for i16 k in [0:K] { + z[k] = 1.0 + } + for i16 k, i16 l in [0:K, 0:K] { + z[k] = z[k] + A_flat[k*K + l] * z[l] + } + ''' + code = _all_code(_lower(_kernel(body))) + assert '@increment_dsd_offset(' not in code + + +def test_no_vectorization_for_nonaffine_index(): + body = ''' + for i16 k in [0:K] { + z[k] = 0.0 + } + for i16 k, i16 l in [0:K, 0:K] { + z[k] = z[k] + A_flat[k*l] * x[l] + } + ''' + code = _all_code(_lower(_kernel(body))) + assert '@increment_dsd_offset(' not in code + + +def test_no_vectorization_when_accumulator_differs(): + body = ''' + for i16 k in [0:K] { + z[k] = 0.0 + } + for i16 k, i16 l in [0:K, 0:K] { + z[k] = x[k] + A_flat[k*K + l] * x[l] + } + ''' + code = _all_code(_lower(_kernel(body))) + assert '@increment_dsd_offset(' not in code From 36da53df08d7f7447d373fd29e09eb6d24a273f8 Mon Sep 17 00:00:00 2001 From: wcy Date: Wed, 2 Sep 2026 01:45:40 +0900 Subject: [PATCH 2/6] Generalize MAC vectorization to accept either loop-variable order Detect the row variable (k, appears alone with coefficient 1 in the destination index) and the reduction variable (l, everything else) from the destination index's affine decomposition instead of assuming statement.variables[0]/[1] position. Accepts both `for (k, l) in [...]` and `for (l, k) in [...]` declarations of the same MAC loop; behavior for the existing (k, l) shape is unchanged (same equality check on the affine decomposition, now checked against either candidate name instead of a hardcoded position). Verified manually with an (l, k)-order kernel: emits identical @fmacs/@increment_dsd_offset output to the (k, l) case (ad hoc script, not committed). Full suite: 381 passed, 12 skipped (unchanged from previous commit). --- spada/syntax/csl/statements.py | 38 +++++++++++++++++++++------------- 1 file changed, 24 insertions(+), 14 deletions(-) diff --git a/spada/syntax/csl/statements.py b/spada/syntax/csl/statements.py index d05039db..942bd22e 100644 --- a/spada/syntax/csl/statements.py +++ b/spada/syntax/csl/statements.py @@ -323,10 +323,29 @@ def _try_emit_vectorized_mac(statement: spir.ForStatement, dtypes: dict[spir.Ide return None if len(statement.body) != 1 or not isinstance(statement.body[0], spir.AssignmentStatement): return None - k_var = statement.variables[0].identifier.name - l_var = statement.variables[1].identifier.name + + # Loop-variable roles are detected from the destination index's affine decomposition rather + # than from statement.variables[0]/[1] position, so both declaration orders — for (k, l) and + # for (l, k) in [...] — are accepted: whichever variable appears alone with coefficient 1 in + # the destination index is the row variable (k); the other is the reduction variable (l). + var_names = [v.identifier.name for v in statement.variables] + var_by_name = {v.identifier.name: v for v in statement.variables} + range_by_name = dict(zip(var_names, statement.range_expression)) + + assign = statement.body[0] + dst = assign.destination + dst_idx = _single_index(dst) + if dst_idx is None: + return None + dst_aff = _affine_of(dst_idx, set(var_names)) + row_candidates = [name for name in var_names if dst_aff == ({name: 1}, 0)] + if len(row_candidates) != 1: + return None + k_var = row_candidates[0] + l_var = next(n for n in var_names if n != k_var) + try: - rk, rl = statement.range_expression + rk, rl = range_by_name[k_var], range_by_name[l_var] k0 = rk.start.eval() if rk.start is not None else 0 kk = rk.stop.eval() ks = rk.step.eval() if rk.step is not None else 1 @@ -338,15 +357,6 @@ def _try_emit_vectorized_mac(statement: spir.ForStatement, dtypes: dict[spir.Ide if not all(isinstance(v, int) for v in (k0, kk, ks, l0, ll, ls)) or (k0, ks, ls) != (0, 1, 1): return None - assign = statement.body[0] - dst = assign.destination - dst_idx = _single_index(dst) - if dst_idx is None: - return None - dst_aff = _affine_of(dst_idx, {k_var}) - if dst_aff != ({k_var: 1}, 0): - return None - src = assign.source.value if isinstance(src, spir.MultiplyAccumulateOperator): acc, mul_b, mul_c = src.a.value, src.b.value, src.c.value @@ -421,8 +431,8 @@ def base_f32(name_node): header_code.write( f'const {dst_dsd} = @get_dsd(mem1d_dsd, .{{ .tensor_access = |__index|{{{kk}}} -> {z_name}[__index] }});\n' f'const {src_dsd} = @get_dsd(mem1d_dsd, .{{ .tensor_access = |__index|{{{kk}}} -> {a_name}[{base_expr}] }});\n') - off = f'{name_to_csl(statement.variables[1].identifier)}' + (f' * {cl}' if cl != 1 else '') - lname = name_to_csl(statement.variables[1].identifier) + off = f'{name_to_csl(var_by_name[l_var].identifier)}' + (f' * {cl}' if cl != 1 else '') + lname = name_to_csl(var_by_name[l_var].identifier) return ( f'// [P4] vectorized MAC: {z_name}[k] += {a_name}[k*{ck}+{lname}*{cl}+{c0}] * {name_to_csl(vec_node.array)}[{x_l}]\n' f'for (@range(i16, {l0}, {ll}, 1)) |{lname}| {{\n' From 42fa1067fb19a2d60ec17c920334909430c0dcf5 Mon Sep 17 00:00:00 2001 From: wcy Date: Wed, 2 Sep 2026 01:51:34 +0900 Subject: [PATCH 3/6] Generalize MAC vectorization dtype dispatch beyond f32 Replace the ad hoc base_f32() gate with a dispatch mirroring FMADSDOp._as_csl (dsd_ops.py) exactly, reusing dsd_ops._get_base_dtype for dtype resolution instead of a hand-rolled base_type/element_type lookup: - f16 matrix/vector operands, f16 accumulator -> @fmach - f32 matrix/vector operands, f32 accumulator -> @fmacs (unchanged) - f32 matrix/vector operands, f16 accumulator -> @fmachs (mixed) The @increment_dsd_offset elem_type argument (previously hardcoded `f32`) and the emitted op name are now both derived from the resolved matrix/vector dtype and dst dtype respectively, via dtype_as_csl(). f16 multiplicands with an f32 accumulator, and all integer dtypes, are intentionally NOT vectorized: FMADSDOp itself has no dispatch branch for f16-mul/f32-accumulate, and DSD_ASSIGNMENT_MAPPING has no integer MAC builtin at all (only @fmach/@fmachs/@fmacs map to FMADSDOp) -- these shapes fall through to the scalar loop, matching upstream's own declared support surface rather than inventing new CSL builtins. SDK verification for @increment_dsd_offset's elem_type parameter: docker exec into cs_sdk_run was unavailable (daemon was down; once started, the container/image itself is not present in this environment -- it belongs to a separate measurement rig). Verified instead against the official CSL language reference (sdk.cerebras.ai/csl/language/dsds, current docs matching SDK 2.10.x): "elem_type ... must be an ABI-compatible numeric type (u16, i16, u32, i32, f16, or f32)" -- f16 is explicitly listed as valid. Manually verified all 5 dtype combinations against the real lowering pipeline (ad hoc script, not committed): all-f16 -> @fmach, all-f32 -> @fmacs, f32-mul/f16-acc -> @fmachs, f16-mul/f32-acc -> correctly NOT vectorized, all-i16 -> correctly NOT vectorized. Full suite: 381 passed, 12 skipped (unchanged). --- spada/syntax/csl/statements.py | 44 +++++++++++++++++++++++++++------- 1 file changed, 36 insertions(+), 8 deletions(-) diff --git a/spada/syntax/csl/statements.py b/spada/syntax/csl/statements.py index 942bd22e..157df4c0 100644 --- a/spada/syntax/csl/statements.py +++ b/spada/syntax/csl/statements.py @@ -317,6 +317,27 @@ def _single_index(node): return idx +def _mac_vectorization_op(mat_vec_dtype: spir.ScalarType, dst_dtype: spir.ScalarType) -> Optional[str]: + """ + Selects the @fmac* builtin for the vectorized MAC idiom, mirroring FMADSDOp._as_csl's dtype + dispatch (dsd_ops.py) exactly: the two multiplicands (mat, vec) must share a dtype + (``mat_vec_dtype``, checked by the caller), paired against the accumulator dtype + (``dst_dtype`` -- here always equal to the destination dtype, since Z[k] is both the + accumulator and the destination in this idiom). + + No integer MAC builtin is present in DSD_ASSIGNMENT_MAPPING -- only @fmach, @fmachs and + @fmacs map to FMADSDOp there -- so integer dtypes intentionally return None: MAC + vectorization stays float-only (f16/f32), matching what FMADSDOp itself supports. + """ + if mat_vec_dtype == spir.ScalarType.f16 and dst_dtype == spir.ScalarType.f16: + return '@fmach' + if mat_vec_dtype == spir.ScalarType.f32 and dst_dtype == spir.ScalarType.f16: + return '@fmachs' + if mat_vec_dtype == spir.ScalarType.f32 and dst_dtype == spir.ScalarType.f32: + return '@fmacs' + return None + + def _try_emit_vectorized_mac(statement: spir.ForStatement, dtypes: dict[spir.Identifier, spir.IRType], header_code: StringIO): if len(statement.variables) != 2 or len(statement.range_expression) != 2: @@ -412,13 +433,20 @@ def classify(node): if mat_node.array.name == dst.array.name or vec_node.array.name == dst.array.name: return None - # f32 전용 (@fmacs) - def base_f32(name_node): - t = dtypes.get(name_node.array) - base = getattr(t, 'base_type', None) or getattr(t, 'element_type', None) - return base == spir.ScalarType.f32 - if not (base_f32(dst) and base_f32(mat_node) and base_f32(vec_node)): + # Type dispatch mirrors FMADSDOp._as_csl (dsd_ops.py): reuse its dtype resolution so a + # scalar/array/stream-typed identifier resolves to the same base ScalarType FMADSDOp itself + # would see. Both multiplicands must share a dtype, and only the f16/f32 combinations + # FMADSDOp models (@fmach, @fmachs, @fmacs) are supported -- integer types are out of scope + # because no integer MAC builtin exists in DSD_ASSIGNMENT_MAPPING. + mat_dtype = dsd_ops._get_base_dtype(dtypes, mat_node) + vec_dtype = dsd_ops._get_base_dtype(dtypes, vec_node) + dst_dtype = dsd_ops._get_base_dtype(dtypes, dst) + if mat_dtype != vec_dtype: + return None + mac_op = _mac_vectorization_op(mat_dtype, dst_dtype) + if mac_op is None: return None + elem_type = dtype_as_csl(mat_dtype) uid = _VEC_COUNTER[0] _VEC_COUNTER[0] += 1 @@ -436,8 +464,8 @@ def base_f32(name_node): return ( f'// [P4] vectorized MAC: {z_name}[k] += {a_name}[k*{ck}+{lname}*{cl}+{c0}] * {name_to_csl(vec_node.array)}[{x_l}]\n' f'for (@range(i16, {l0}, {ll}, 1)) |{lname}| {{\n' - f' const __vecmac_a_{uid} = @increment_dsd_offset({src_dsd}, {off}, f32);\n' - f' @fmacs({dst_dsd}, {dst_dsd}, __vecmac_a_{uid}, {name_to_csl(vec_node.array)}[{x_l}]);\n' + f' const __vecmac_a_{uid} = @increment_dsd_offset({src_dsd}, {off}, {elem_type});\n' + f' {mac_op}({dst_dsd}, {dst_dsd}, __vecmac_a_{uid}, {name_to_csl(vec_node.array)}[{x_l}]);\n' f'}}\n') From c14921401410ef264f8f710491d40143d259c8f4 Mon Sep 17 00:00:00 2001 From: wcy Date: Wed, 2 Sep 2026 01:54:12 +0900 Subject: [PATCH 4/6] Fold MAC vectorization into --disable-dsd; remove the dedicated flag Per issue #69 ask 5 (maintainer: "this generalizes DSD detection to ND loops" -- no separate flag needed), remove --disable-mac-vectorization entirely: - spada/cli/compiler.py: drop the click option, the generate_program()/compile_spatial_ir() parameter, and the pass-through call argument. - spada/cli/appliance_compiler.py: same, for compile_spatial_ir_appliance() (this entry point postdates P4 -- it never had the flag to begin with, so this is purely keeping it in sync with compiler.py's shared codegen_options). - spada/lowering/spatial_ir_to_csl.py: drop disable_mac_vectorization from lower_spatial_ir_to_csl() and generate_rectangle(), and the `cslstmt.DISABLE_MAC_VECTORIZATION = True` assignment. - spada/syntax/csl/statements.py: drop the DISABLE_MAC_VECTORIZATION module global; emit_for's gate is now `if not dsd_ops.DISABLE_DSD:` only, with no second condition. Also rewrote the pass's header comment to reflect the current (order- and dtype-generalized, still 2-level-only) scope and record the @increment_dsd_offset f16 verification. - tests/spatial_ir/test_mac_vectorization.py: rewrote test_scalar_fallback_when_disabled to pass disable_dsd=True (was disable_mac_vectorization=True) and reset dsd_ops.DISABLE_DSD (was cslstmt.DISABLE_MAC_VECTORIZATION) in its cleanup -- both are sticky module-level globals, so the reset target had to change along with the flag. Semantic note: this was already true in practice before this commit -- emit_for's old gate was `if not dsd_ops.DISABLE_DSD and not DISABLE_MAC_VECTORIZATION`, so --disable-dsd already disabled MAC vectorization too. What's removed is the ability to disable *only* MAC vectorization while leaving other DSD ops enabled; no prior behavior for --disable-dsd users changes. Verified both CLI entry points still parse correctly (`python -m spada.cli.compiler --help` / `python -m spada.cli.appliance_compiler --help`) with no --disable-mac-vectorization option and no crash. Full suite: 381 passed, 12 skipped (unchanged). --- spada/cli/appliance_compiler.py | 2 -- spada/cli/compiler.py | 7 +------ spada/lowering/spatial_ir_to_csl.py | 8 ++------ spada/syntax/csl/statements.py | 16 ++++++++++------ tests/spatial_ir/test_mac_vectorization.py | 16 ++++++++++------ 5 files changed, 23 insertions(+), 26 deletions(-) diff --git a/spada/cli/appliance_compiler.py b/spada/cli/appliance_compiler.py index b43ecd37..f35037e9 100644 --- a/spada/cli/appliance_compiler.py +++ b/spada/cli/appliance_compiler.py @@ -92,7 +92,6 @@ def compile_spatial_ir_appliance(input_file: str, output_folder: str, param: lis generate_only: bool, simulator: bool, mgmt_namespace: Optional[str], resource_cpu: Optional[int], resource_mem: Optional[int], check_version: bool, disable_benchmarking: bool, disable_asynchronous: bool, disable_dsd: bool, - disable_mac_vectorization: bool, disable_map: bool, disable_task_fusion: bool, disable_task_recycling: bool, disable_copy_elision: bool, disable_close_elision: bool, disable_switching: bool): program = generate_program( @@ -104,7 +103,6 @@ def compile_spatial_ir_appliance(input_file: str, output_folder: str, param: lis disable_benchmarking=disable_benchmarking, disable_asynchronous=disable_asynchronous, disable_dsd=disable_dsd, - disable_mac_vectorization=disable_mac_vectorization, disable_map=disable_map, disable_task_fusion=disable_task_fusion, disable_task_recycling=disable_task_recycling, diff --git a/spada/cli/compiler.py b/spada/cli/compiler.py index 723e53ab..aa87019d 100644 --- a/spada/cli/compiler.py +++ b/spada/cli/compiler.py @@ -62,8 +62,6 @@ def codegen_options(func): help='Disable benchmarking code generation (and memory overhead)'), click.option('--disable-asynchronous', is_flag=True, help='Disable asynchronous task code generation'), click.option('--disable-dsd', is_flag=True, help='Disable DSD operation detection and code generation'), - click.option('--disable-mac-vectorization', is_flag=True, - help='Disable vectorization of nested multiply-accumulate loops into @fmac* DSD operations'), click.option('--disable-map', is_flag=True, help='Disable @map operation detection and code generation'), click.option('--disable-task-fusion', is_flag=True, help='Disable task fusion optimization'), click.option('--disable-task-recycling', is_flag=True, help='Disable task ID recycling'), @@ -78,7 +76,6 @@ def codegen_options(func): def generate_program(input_file: str, output_folder: str, param: list[str] = (), offset_x: int = 0, offset_y: int = 0, disable_benchmarking: bool = False, disable_asynchronous: bool = False, disable_dsd: bool = False, - disable_mac_vectorization: bool = False, disable_map: bool = False, disable_task_fusion: bool = False, disable_task_recycling: bool = False, disable_copy_elision: bool = False, disable_close_elision: bool = False, disable_switching: bool = False) -> GeneratedProgram: @@ -141,7 +138,6 @@ def generate_program(input_file: str, output_folder: str, param: list[str] = (), disable_benchmarking=disable_benchmarking, disable_asynchronous=disable_asynchronous, disable_dsd=disable_dsd, - disable_mac_vectorization=disable_mac_vectorization, task_fusion=not disable_task_fusion, copy_elision=not disable_copy_elision, task_id_recycling=not disable_task_recycling, @@ -260,7 +256,7 @@ def cslc_arguments(program: GeneratedProgram, hardware_fabric: Optional[bool] = @click.option('--generate-only', '-g', is_flag=True, help='Only generate the output files without compiling them') def compile_spatial_ir(input_file: str, output_folder: str, param: list[str], offset_x: int, offset_y: int, generate_only: bool, disable_benchmarking: bool, - disable_asynchronous: bool, disable_dsd: bool, disable_mac_vectorization: bool, + disable_asynchronous: bool, disable_dsd: bool, disable_map: bool, disable_task_fusion: bool, disable_task_recycling: bool, disable_copy_elision: bool, disable_close_elision: bool, disable_switching: bool): @@ -273,7 +269,6 @@ def compile_spatial_ir(input_file: str, output_folder: str, param: list[str], of disable_benchmarking=disable_benchmarking, disable_asynchronous=disable_asynchronous, disable_dsd=disable_dsd, - disable_mac_vectorization=disable_mac_vectorization, disable_map=disable_map, disable_task_fusion=disable_task_fusion, disable_task_recycling=disable_task_recycling, diff --git a/spada/lowering/spatial_ir_to_csl.py b/spada/lowering/spatial_ir_to_csl.py index cadb29ce..4c15a246 100644 --- a/spada/lowering/spatial_ir_to_csl.py +++ b/spada/lowering/spatial_ir_to_csl.py @@ -54,7 +54,6 @@ def lower_spatial_ir_to_csl(kernel: spir.Kernel, disable_benchmarking: bool = False, disable_asynchronous: bool = False, disable_dsd: bool = False, - disable_mac_vectorization: bool = False, task_fusion: bool = True, copy_elision: bool = True, prune_memory: bool = True, @@ -159,7 +158,7 @@ def lower_spatial_ir_to_csl(kernel: spir.Kernel, rect_code, color_map = generate_rectangle(kernel, rect, routing_instructions, scalar_arguments, use_memcpy_mode, stream_rects, channel_to_color, disable_benchmarking, disable_asynchronous, disable_dsd, task_fusion, - task_id_recycling, disable_mac_vectorization) + task_id_recycling) color_maps.append(color_map) csl_codes.append(CodeFile(csl_name, rect_code)) @@ -319,8 +318,7 @@ def generate_rectangle(kernel: spir.Kernel, disable_asynchronous: bool = False, disable_dsd: bool = False, task_fusion: bool = True, - task_id_recycling: bool = True, - disable_mac_vectorization: bool = False) -> tuple[str, dict[str, int]]: + task_id_recycling: bool = True) -> tuple[str, dict[str, int]]: # Code generation carets header = StringIO() current_code = StringIO() @@ -328,8 +326,6 @@ def generate_rectangle(kernel: spir.Kernel, if disable_dsd: dsd_ops.DISABLE_DSD = True - if disable_mac_vectorization: - cslstmt.DISABLE_MAC_VECTORIZATION = True header.write(""" param memcpy_params; diff --git a/spada/syntax/csl/statements.py b/spada/syntax/csl/statements.py index 157df4c0..5b211199 100644 --- a/spada/syntax/csl/statements.py +++ b/spada/syntax/csl/statements.py @@ -252,11 +252,15 @@ def emit_assignment(statement: spir.AssignmentStatement, dsds: UniqueDSDDict, dt # [P4] Targeted vectorization of the nested multiply-accumulate loop # for (k, l) in [0:Kk, 0:Kl]: Z[k] = Z[k] + A[k*Ck + l*Cl + C0] * X[f(l)] -# into the handwritten CSL idiom (strided base DSD + per-l @increment_dsd_offset + @fmacs). -# Conservative: fires only on the exact f32 MAC shape with Z distinct from A and X; -# anything else falls through to scalar loops. @increment_dsd_offset is available from -# SDK 1.x (used by Cerebras' own csl-examples v1.4.0 cholesky benchmark). -DISABLE_MAC_VECTORIZATION = False +# (declared in either variable order) into the handwritten CSL idiom (strided base DSD + per-l +# @increment_dsd_offset + one of @fmach/@fmachs/@fmacs, matching FMADSDOp's own f16/f32 dtype +# dispatch). Conservative: fires only on this exact 2-level MAC shape with Z distinct from A and +# X; anything else falls through to scalar loops. @increment_dsd_offset is available from SDK 1.x +# (used by Cerebras' own csl-examples v1.4.0 cholesky benchmark) and its elem_type parameter is +# documented to accept f16 as well as f32 (sdk.cerebras.ai/csl/language/dsds). Generalized +# ND-loop / any-DSD-op vectorization (issue #69 asks 1 and 4) is out of scope here -- only the +# loop-variable order (ask 2), dtype restriction (ask 3), and the --disable-dsd fold-in (ask 5) +# are addressed; this pass is gated solely by dsd_ops.DISABLE_DSD, with no separate flag. _VEC_COUNTER = [0] @@ -480,7 +484,7 @@ def emit_for(statement: spir.ForStatement, dsds: UniqueDSDDict, dtypes: dict[spi :param header_code: The header code to include. :return: The generated CSL for loop statement. """ - if not dsd_ops.DISABLE_DSD and not DISABLE_MAC_VECTORIZATION: + if not dsd_ops.DISABLE_DSD: vectorized = _try_emit_vectorized_mac(statement, dtypes, header_code) if vectorized is not None: return vectorized diff --git a/tests/spatial_ir/test_mac_vectorization.py b/tests/spatial_ir/test_mac_vectorization.py index ffd96266..33c6e248 100644 --- a/tests/spatial_ir/test_mac_vectorization.py +++ b/tests/spatial_ir/test_mac_vectorization.py @@ -5,13 +5,15 @@ for (k, l) in [0:K, 0:K]: z[k] = z[k] + A[k*K + l] * x[l] -into a strided base DSD plus a per-column ``@increment_dsd_offset`` and ``@fmacs``, -and must fall back to scalar loops for every shape it cannot prove safe -(aliasing, non-affine indices, non-f32 dtypes, or the explicit disable flag). +into a strided base DSD plus a per-column ``@increment_dsd_offset`` and one of +``@fmach``/``@fmachs``/``@fmacs`` (dtype-dependent, mirroring FMADSDOp), and must fall back to +scalar loops for every shape it cannot prove safe (aliasing, non-affine indices, unsupported +dtype combinations, or ``--disable-dsd``). There is no separate disable flag for this pass: it is +gated solely by ``dsd_ops.DISABLE_DSD``, same as every other DSD operation. """ import pytest from spada.lowering.spatial_ir_to_csl import lower_spatial_ir_to_csl -from spada.syntax.csl import statements as cslstmt +from spada.syntax.csl import dsd_ops from spada.syntax.spatial_ir import parser, passes @@ -70,11 +72,13 @@ def test_mac_loop_is_vectorized(): def test_scalar_fallback_when_disabled(): + # MAC vectorization has no dedicated disable flag -- it is folded into --disable-dsd, the + # same switch that disables every other DSD operation (issue #69 ask 5). try: - code = _all_code(_lower(_kernel(MAC_BODY), disable_mac_vectorization=True)) + code = _all_code(_lower(_kernel(MAC_BODY), disable_dsd=True)) assert '@increment_dsd_offset(' not in code finally: - cslstmt.DISABLE_MAC_VECTORIZATION = False # module flag is sticky, reset for other tests + dsd_ops.DISABLE_DSD = False # module flag is sticky, reset for other tests def test_no_vectorization_when_accumulator_aliases_matrix(): From 803a8744ea14393bbd045d705081ed717a7c4dc1 Mon Sep 17 00:00:00 2001 From: wcy Date: Wed, 2 Sep 2026 01:56:50 +0900 Subject: [PATCH 5/6] Expand MAC vectorization tests across loop order and dtype Extends tests/spatial_ir/test_mac_vectorization.py from 6 to 20 tests to cover the (2) order and (3) dtype generalizations: - test_mac_loop_vectorized_for_order_and_dtype: positive matrix, {kl, lk} x {(f32,f32)->@fmacs, (f16,f16)->@fmach, (f32 mul,f16 acc)->@fmachs} = 6 cases. - test_no_vectorization_for_unsupported_dtype_combo: {kl, lk} x {(f16 mul, f32 acc) -- no FMADSDOp branch, (i16, i16) -- no integer @fmac* builtin} = 4 cases. - The four pre-existing negative tests (accumulator aliases matrix, accumulator aliases vector, non-affine index, accumulator differs from destination) are now parametrized over {kl, lk} = 8 cases, up from 4, to confirm the aliasing/affine/accumulator guards still apply correctly regardless of which position declares the row vs. reduction variable. _kernel() gained mat_ty/out_ty parameters (defaulting to the original all-f32 behavior) so stream types can vary with the tested dtype combination -- receive/send would otherwise reject a dtype mismatch before MAC vectorization is ever reached. Armed per art.14 (verified by temporarily breaking the guard under test and confirming the corresponding new cases go red, then reverting): - Aliasing guard disabled -> all 4 test_no_vectorization_when_accumulator_aliases_{matrix,vector}[kl|lk] cases fail as expected (vectorization no longer suppressed). - Dtype gate bypassed (falling back to @fmacs unconditionally) -> all 4 test_no_vectorization_for_unsupported_dtype_combo[...] cases fail as expected. Both reverted; suite is green again with the real guards in place. Full suite: 395 passed, 12 skipped (baseline 375 + 20 in this file, 0 regressions). --- tests/spatial_ir/test_mac_vectorization.py | 125 +++++++++++++++------ 1 file changed, 93 insertions(+), 32 deletions(-) diff --git a/tests/spatial_ir/test_mac_vectorization.py b/tests/spatial_ir/test_mac_vectorization.py index 33c6e248..6330ba06 100644 --- a/tests/spatial_ir/test_mac_vectorization.py +++ b/tests/spatial_ir/test_mac_vectorization.py @@ -17,17 +17,22 @@ from spada.syntax.spatial_ir import parser, passes -def _kernel(body: str, decls: str = None) -> str: - decls = decls if decls is not None else ''' - f32[K*K] A_flat - f32[K] x - f32[K] z +def _kernel(body: str, decls: str = None, mat_ty: str = 'f32', out_ty: str = None) -> str: + # mat_ty covers A_flat/x (the two @fmac* multiplicands, which must share a dtype); out_ty + # covers z (the accumulator/destination), defaulting to mat_ty. Stream types must match the + # local arrays they feed (receive/send are otherwise rejected as a dtype mismatch upstream of + # MAC vectorization entirely), so both need to vary together with the local declarations. + out_ty = out_ty if out_ty is not None else mat_ty + decls = decls if decls is not None else f''' + {mat_ty}[K*K] A_flat + {mat_ty}[K] x + {out_ty}[K] z ''' return f''' kernel @t( - stream[2, 2] readonly inp_A, - stream[2, 2] readonly inp_x, - stream[2, 2] writeonly out + stream<{mat_ty}, K*K>[2, 2] readonly inp_A, + stream<{mat_ty}, K>[2, 2] readonly inp_x, + stream<{out_ty}, K>[2, 2] writeonly out ) {{ place i16 i, i16 j in [0:2, 0:2] {{ {decls} @@ -81,54 +86,110 @@ def test_scalar_fallback_when_disabled(): dsd_ops.DISABLE_DSD = False # module flag is sticky, reset for other tests -def test_no_vectorization_when_accumulator_aliases_matrix(): +# --- Loop-variable-order matrix (issue #69 ask 2: (k, l) and (l, k) must both be detected) ----- + +def _order_decl(order: str) -> str: + """The `for in [0:K, 0:K]` variable declaration for a given loop-variable order.""" + return {'kl': 'i16 k, i16 l', 'lk': 'i16 l, i16 k'}[order] + + +def _mac_body(order: str) -> str: + return f''' + for i16 k in [0:K] {{ + z[k] = 0.0 + }} + for {_order_decl(order)} in [0:K, 0:K] {{ + z[k] = z[k] + A_flat[k*K + l] * x[l] + }} + ''' + + +ORDERS = ['kl', 'lk'] + +# (mat_ty, out_ty, expected op) for the dtype combinations FMADSDOp actually models +# (dsd_ops.DSD_ASSIGNMENT_MAPPING has no integer @fmac* builtin, so only f16/f32 are covered). +SUPPORTED_DTYPE_COMBOS = [ + ('f32', 'f32', '@fmacs('), # both multiplicands and accumulator f32 + ('f16', 'f16', '@fmach('), # both multiplicands and accumulator f16 + ('f32', 'f16', '@fmachs('), # f32 multiplicands, f16 accumulate (mixed) +] + +# (mat_ty, out_ty) combinations that must NOT vectorize: either FMADSDOp itself has no dispatch +# branch for the combination (f16 multiplicands / f32 accumulate), or no @fmac* builtin exists +# for the dtype at all (integer). +UNSUPPORTED_DTYPE_COMBOS = [ + ('f16', 'f32'), + ('i16', 'i16'), +] + + +@pytest.mark.parametrize('order', ORDERS) +@pytest.mark.parametrize('mat_ty, out_ty, expected_op', SUPPORTED_DTYPE_COMBOS) +def test_mac_loop_vectorized_for_order_and_dtype(order, mat_ty, out_ty, expected_op): + code = _all_code(_lower(_kernel(_mac_body(order), mat_ty=mat_ty, out_ty=out_ty))) + assert '@increment_dsd_offset(' in code + assert expected_op in code + + +@pytest.mark.parametrize('order', ORDERS) +@pytest.mark.parametrize('mat_ty, out_ty', UNSUPPORTED_DTYPE_COMBOS) +def test_no_vectorization_for_unsupported_dtype_combo(order, mat_ty, out_ty): + code = _all_code(_lower(_kernel(_mac_body(order), mat_ty=mat_ty, out_ty=out_ty))) + assert '@increment_dsd_offset(' not in code + + +@pytest.mark.parametrize('order', ORDERS) +def test_no_vectorization_when_accumulator_aliases_matrix(order): # z appears as the matrix operand: DSD reordering would change semantics. - body = ''' - for i16 k in [0:K] { + body = f''' + for i16 k in [0:K] {{ z[k] = 1.0 - } - for i16 k, i16 l in [0:K, 0:K] { + }} + for {_order_decl(order)} in [0:K, 0:K] {{ z[k] = z[k] + z[k*1 + l*0] * x[l] - } + }} ''' code = _all_code(_lower(_kernel(body))) assert '@increment_dsd_offset(' not in code -def test_no_vectorization_when_accumulator_aliases_vector(): - body = ''' - for i16 k in [0:K] { +@pytest.mark.parametrize('order', ORDERS) +def test_no_vectorization_when_accumulator_aliases_vector(order): + body = f''' + for i16 k in [0:K] {{ z[k] = 1.0 - } - for i16 k, i16 l in [0:K, 0:K] { + }} + for {_order_decl(order)} in [0:K, 0:K] {{ z[k] = z[k] + A_flat[k*K + l] * z[l] - } + }} ''' code = _all_code(_lower(_kernel(body))) assert '@increment_dsd_offset(' not in code -def test_no_vectorization_for_nonaffine_index(): - body = ''' - for i16 k in [0:K] { +@pytest.mark.parametrize('order', ORDERS) +def test_no_vectorization_for_nonaffine_index(order): + body = f''' + for i16 k in [0:K] {{ z[k] = 0.0 - } - for i16 k, i16 l in [0:K, 0:K] { + }} + for {_order_decl(order)} in [0:K, 0:K] {{ z[k] = z[k] + A_flat[k*l] * x[l] - } + }} ''' code = _all_code(_lower(_kernel(body))) assert '@increment_dsd_offset(' not in code -def test_no_vectorization_when_accumulator_differs(): - body = ''' - for i16 k in [0:K] { +@pytest.mark.parametrize('order', ORDERS) +def test_no_vectorization_when_accumulator_differs(order): + body = f''' + for i16 k in [0:K] {{ z[k] = 0.0 - } - for i16 k, i16 l in [0:K, 0:K] { + }} + for {_order_decl(order)} in [0:K, 0:K] {{ z[k] = x[k] + A_flat[k*K + l] * x[l] - } + }} ''' code = _all_code(_lower(_kernel(body))) assert '@increment_dsd_offset(' not in code From fdef30c0f557db6e3704a6c246caaa3de22456af Mon Sep 17 00:00:00 2001 From: wcy Date: Wed, 2 Sep 2026 01:59:43 +0900 Subject: [PATCH 6/6] Cosmetic: rejoin compiler.py signature line Leftover line wrap from adding then removing disable_mac_vectorization across the earlier P4-rebase/flag-removal commits; no functional change. Rejoins compile_spatial_ir's signature to match upstream's original formatting so the final diff against eaf8d1c is minimal. --- spada/cli/compiler.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/spada/cli/compiler.py b/spada/cli/compiler.py index aa87019d..46785c46 100644 --- a/spada/cli/compiler.py +++ b/spada/cli/compiler.py @@ -256,8 +256,7 @@ def cslc_arguments(program: GeneratedProgram, hardware_fabric: Optional[bool] = @click.option('--generate-only', '-g', is_flag=True, help='Only generate the output files without compiling them') def compile_spatial_ir(input_file: str, output_folder: str, param: list[str], offset_x: int, offset_y: int, generate_only: bool, disable_benchmarking: bool, - disable_asynchronous: bool, disable_dsd: bool, - disable_map: bool, + disable_asynchronous: bool, disable_dsd: bool, disable_map: bool, disable_task_fusion: bool, disable_task_recycling: bool, disable_copy_elision: bool, disable_close_elision: bool, disable_switching: bool): program = generate_program(