From c58bb0652073236f1fd9003cf1480d8217862cff Mon Sep 17 00:00:00 2001 From: RJ Ascani Date: Thu, 10 Sep 2026 16:47:48 -0700 Subject: [PATCH] [Cortex-M] Work around transpose convolution scratch sizing Reserve the larger of the CMSIS-NN reported size and the rolling-buffer kernel requirement during export, leaving reverse-convolution allocations unchanged. Allow the larger allocation in the optional runtime check while awaiting https://github.com/ARM-software/CMSIS-NN/pull/243. Enable runtime checks in the test runner and cover strided pointwise transpose convolution in explicit layout, with and without Hardtanh. The focused M55 suite passed with 51 passes and 9 expected failures; 30 separate allocation checks also passed. Authored with Codex. --- .../ops/op_quantized_transpose_conv2d.cpp | 5 ++++- .../cortex_m/passes/scratch_buffer_sizes.py | 16 ++++++++++++++-- backends/cortex_m/test/build_test_runner.sh | 2 +- .../test/test_explicit_layout_pipeline.py | 18 ++++++++++++++++++ 4 files changed, 37 insertions(+), 4 deletions(-) diff --git a/backends/cortex_m/ops/op_quantized_transpose_conv2d.cpp b/backends/cortex_m/ops/op_quantized_transpose_conv2d.cpp index 4ac9b2338e6..aec006e28cd 100644 --- a/backends/cortex_m/ops/op_quantized_transpose_conv2d.cpp +++ b/backends/cortex_m/ops/op_quantized_transpose_conv2d.cpp @@ -213,7 +213,10 @@ static Tensor& quantized_transpose_conv2d_out_impl( #ifdef CORTEX_M_ENABLE_RUNTIME_CHECKS const int32_t buffer_bytes = arm_transpose_conv_s8_get_buffer_size( &transpose_conv_params, &input_dims, &filter_dims, &output_dims); - if (scratch.nbytes() != static_cast(buffer_bytes)) { + // AOT reserves max(API size, corrected kernel size), so it may be larger. + // TODO: Restore equality once our CMSIS-NN pin includes + // https://github.com/ARM-software/CMSIS-NN/pull/243. + if (scratch.nbytes() < static_cast(buffer_bytes)) { ET_LOG( Error, "quantized_transpose_conv2d_out: scratch buffer size incorrect - actual: (%d) needed: (%d)", diff --git a/backends/cortex_m/passes/scratch_buffer_sizes.py b/backends/cortex_m/passes/scratch_buffer_sizes.py index 65a3a178757..e6c8b2b8454 100644 --- a/backends/cortex_m/passes/scratch_buffer_sizes.py +++ b/backends/cortex_m/passes/scratch_buffer_sizes.py @@ -219,7 +219,7 @@ def cmsis_nn_transpose_conv_buffer_size( filter_nhwc = [c_out, kernel_h, kernel_w, kernel_c_in] padding_offsets_hw = [int(output_padding[0]), int(output_padding[1])] - return [ + buffer_bytes, reverse_buffer_bytes = ( int( cmsis_nn.transpose_conv_buffer_size( backend, @@ -253,7 +253,19 @@ def cmsis_nn_transpose_conv_buffer_size( activation_max=output_qmax, ) ), - ] + ) + + # A zero reverse buffer means CMSIS-NN selected its rolling-buffer path. + if reverse_buffer_bytes == 0: + # The sizing API uses stride.h where the kernel uses stride.w. Preserve + # the API's requirement while ensuring enough space for the kernel. + # TODO: Remove this correction once our CMSIS-NN pin includes + # https://github.com/ARM-software/CMSIS-NN/pull/243. + buffer_w = (input_nhwc[2] - 1) * stride_hw[1] + max(kernel_w, stride_hw[1]) + buffer_h = max(kernel_h, stride_hw[0]) + buffer_bytes = max(buffer_bytes, buffer_w * buffer_h * c_out * 4) + + return [buffer_bytes, reverse_buffer_bytes] def cmsis_nn_avgpool_buffer_size( diff --git a/backends/cortex_m/test/build_test_runner.sh b/backends/cortex_m/test/build_test_runner.sh index 39a760ee39f..49d376bd0b4 100755 --- a/backends/cortex_m/test/build_test_runner.sh +++ b/backends/cortex_m/test/build_test_runner.sh @@ -80,4 +80,4 @@ ops_list=( cortex_m::quantized_batch_matmul.out ) -${build_executor_runner} --pte=semihosting --bundleio --target="${target}" --output="${build_root_test_dir}" --select_ops_list="$(join_by_comma "${ops_list[@]}")" --extra_build_flags="-DET_ATOL=5.0 -DET_RTOL=1.0 -DET_ARM_BAREMETAL_SCRATCH_TEMP_ALLOCATOR_POOL_SIZE=0" +${build_executor_runner} --pte=semihosting --bundleio --target="${target}" --output="${build_root_test_dir}" --select_ops_list="$(join_by_comma "${ops_list[@]}")" --extra_build_flags="-DCORTEX_M_ENABLE_RUNTIME_CHECKS=ON -DET_ATOL=5.0 -DET_RTOL=1.0 -DET_ARM_BAREMETAL_SCRATCH_TEMP_ALLOCATOR_POOL_SIZE=0" diff --git a/backends/cortex_m/test/test_explicit_layout_pipeline.py b/backends/cortex_m/test/test_explicit_layout_pipeline.py index 7c93233e768..a278209f658 100644 --- a/backends/cortex_m/test/test_explicit_layout_pipeline.py +++ b/backends/cortex_m/test/test_explicit_layout_pipeline.py @@ -148,6 +148,24 @@ def test_explicit_layout_reuses_pad(): assert _count(program, exir_ops.edge.cortex_m.pad.default) == 1 +@pytest.mark.parametrize("hardtanh", [False, True]) +def test_implementation_transpose_conv2d_strided_pointwise(hardtanh): + torch.manual_seed(0) + inputs = (torch.randn(2, 4, 9).unsqueeze(2),) + model = torch.nn.Sequential( + torch.nn.ConvTranspose2d(4, 4, (1, 1), stride=(1, 2)), + torch.nn.Hardtanh(-0.5, 0.5) if hardtanh else torch.nn.Identity(), + ).eval() + tester = _run_explicit_layout_passes(CortexMTester(model, inputs)) + program = tester.get_artifact(StageType.RUN_PASSES).exported_program() + assert ( + _count(program, exir_ops.edge.cortex_m.quantized_transpose_conv2d_nhwc.default) + == 1 + ) + tester.to_executorch().serialize() + tester.run_method_and_compare_outputs(inputs=inputs, qtol=1) + + def test_explicit_layout_rejects_unsupported_spatial_operator(): tester = CortexMTester(UnsupportedAvgPool(), (torch.randn(1, 3, 8, 8),))