From 28befbdf4b94d616afcc8f0b441159b15753cdaf Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Sun, 27 Sep 2026 22:59:03 +0000 Subject: [PATCH 1/7] bench: a hexagon key for the instruction-count ratchet The srdif engine (tap/DspTap#42) met "must not regress" on every key this ratchet measured, but none of them is Hexagon, the third target of MuTap, DspTap's main consumer: MuTap's Hexagon ratchet read +12 ... +13 % on its chain workloads at the bump to 72977aa. This adds the target so DspTap gates it directly: - cmake/hexagon-linux-musl.cmake, adapted from MuTap's (the same CodeLinaro clang 19.1.5 toolchain, -mv68 -mhvx -mhvx-length=128b, static musl), so the two repositories' Hexagon counts are comparable. - scripts/icount.py: target `hexagon`, run under qemu-hexagon user mode. - bench.yml: a `hexagon` key in the matrix, which builds a plugin-enabled qemu-hexagon from the pinned QEMU 8.2.2 release (cached on its digest), downloads the toolchain, and measures the size probes with llvm-size. As a hosted target it also builds and gates rfft_f64_512. The key has no baselines yet, so its job refuses on a pull request until a workflow_dispatch run on this branch seeds them (bench/README.md, "Seeding"); the .text ceilings are 0 (not recorded) until then. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- .github/workflows/bench.yml | 88 ++++++++++++++++++++++++++++++++-- cmake/hexagon-linux-musl.cmake | 63 ++++++++++++++++++++++++ scripts/icount.py | 12 +++-- 3 files changed, 155 insertions(+), 8 deletions(-) create mode 100644 cmake/hexagon-linux-musl.cmake diff --git a/.github/workflows/bench.yml b/.github/workflows/bench.yml index 8450f7f..349cbe0 100644 --- a/.github/workflows/bench.yml +++ b/.github/workflows/bench.yml @@ -11,6 +11,8 @@ name: Bench # m55 Cortex-M55, Helium (mps3-an547) CMSIS-DSP (deployed profile) # m55-ooura Cortex-M55, TAP_DSP_FFT_CMSIS=OFF (mps3-an547) srdif float32 fallback # (the key's name is historical; baselines are keyed on it) +# hexagon Hexagon v68 + HVX-128, clang 19 (qemu-hexagon user mode) srdif float32 +# and double (a hosted target, so rfft_f64_512 is built and gated too) # # Every key measures what ships through tap::dsp::basic_real_fft: the srdif # engine (fft/srdif.h), or CMSIS-DSP behind the same class on the `m55` key, @@ -53,7 +55,7 @@ on: permissions: contents: read -# The expensive workflow: five QEMU jobs. One run per ref at a time. +# The expensive workflow: six QEMU jobs. One run per ref at a time. concurrency: group: bench-${{ github.workflow }}-${{ github.ref }} cancel-in-progress: true @@ -115,6 +117,20 @@ jobs: text_ceiling_f32: 29056 text_ceiling_q15: 27968 text_ceiling_q31: 27392 + # Hexagon: MuTap's third target (its Hexagon CI leg and ratchet use + # this toolchain, these flags and this QEMU), added when the srdif + # engine was found to cost more instructions there than the engine + # it replaced, which no key here measured (bench/README.md). + # clang, not GCC; hosted (static musl) under qemu-hexagon, which + # the job builds with plugin support. The size probes are measured + # with the toolchain's llvm-size. + - key: hexagon + arch: hexagon + toolchain: cmake/hexagon-linux-musl.cmake + flags: "" + text_ceiling_f32: 0 + text_ceiling_q15: 0 + text_ceiling_q31: 0 env: # Commit the tag pointed at when pinned (tags are movable; commit SHAs # are not); the header's digest is verified on download. This SHA is @@ -125,6 +141,12 @@ jobs: # that is never shipped (tools/qemu_insn_plugin/insn_count.c). QEMU_PLUGIN_HEADER_URL: https://raw.githubusercontent.com/qemu/qemu/11aa0b1ff115b86160c4d37e7c37e6a6b13b77ea/include/qemu/qemu-plugin.h QEMU_PLUGIN_HEADER_SHA256: "c53a2af163e80e3f4bc6c60dbdfc84003db329d757e37cd8a16a77e1d82606ff" + # The hexagon key only: the QEMU release its qemu-hexagon is built + # from (digest-verified, the plugin header's release) and the + # CodeLinaro toolchain (MuTap's pins). + QEMU_SRC_URL: https://download.qemu.org/qemu-8.2.2.tar.xz + QEMU_SRC_SHA256: 847346c1b82c1a54b2c38f6edbd85549edeb17430b7d4d3da12620e2962bc4f3 + HEXAGON_TOOLCHAIN: clang+llvm-19.1.5-cross-hexagon-unknown-linux-musl PLUGIN: /tmp/libinsncount.so BUILD_DIR: build-${{ matrix.key }} SIZE_DIR: build-${{ matrix.key }}-minsizerel @@ -133,18 +155,72 @@ jobs: - uses: actions/checkout@v4 - name: Install Arm toolchain and QEMU + if: matrix.arch != 'hexagon' run: > sudo apt-get update -q && sudo apt-get install -y -q gcc-arm-none-eabi qemu-system-arm libglib2.0-dev pkg-config + # Neither Debian's nor CodeLinaro's qemu-hexagon enables TCG plugins, + # so the hexagon key builds its own from the pinned release (the same + # pin and digest as MuTap's hexagon job, cached on the digest). + - name: Install Hexagon build dependencies + if: matrix.arch == 'hexagon' + run: > + sudo apt-get update -q && + sudo apt-get install -y -q libglib2.0-dev pkg-config ninja-build flex bison zstd + + - name: Cache plugin-enabled qemu-hexagon + if: matrix.arch == 'hexagon' + id: qemu-hex + uses: actions/cache@v4 + with: + path: ~/qemu-hexagon-plugins + key: qemu-hexagon-plugins-${{ env.QEMU_SRC_SHA256 }}-1 + + - name: Build plugin-enabled qemu-hexagon + if: matrix.arch == 'hexagon' && steps.qemu-hex.outputs.cache-hit != 'true' + run: | + curl -sfLo /tmp/qemu-src.tar.xz "$QEMU_SRC_URL" + actual=$(sha256sum /tmp/qemu-src.tar.xz | cut -d' ' -f1) + if [ "$actual" != "$QEMU_SRC_SHA256" ]; then + echo "::error::qemu source checksum mismatch"; exit 1 + fi + tar -xJf /tmp/qemu-src.tar.xz -C /tmp + cd /tmp/qemu-*/ + ./configure --target-list=hexagon-linux-user --enable-plugins \ + --disable-docs --disable-tools --disable-system + ninja -C build qemu-hexagon + mkdir -p ~/qemu-hexagon-plugins + cp build/qemu-hexagon ~/qemu-hexagon-plugins/ + + - name: Download the Hexagon toolchain + if: matrix.arch == 'hexagon' + run: | + mkdir -p "$RUNNER_TEMP/hexagon-toolchain" + curl -fsSL "https://artifacts.codelinaro.org/artifactory/codelinaro-toolchain-for-hexagon/19.1.5/${HEXAGON_TOOLCHAIN}.tar.zst" \ + | tar --zstd -x -C "$RUNNER_TEMP/hexagon-toolchain" + cxx=$(find "$RUNNER_TEMP/hexagon-toolchain" -maxdepth 5 \( -type f -o -type l \) \ + -name 'hexagon-unknown-linux-musl-clang++' | sort | head -1) + test -n "$cxx" || { echo "::error::no hexagon clang++ in the toolchain archive"; exit 1; } + echo "HEXAGON_TOOLCHAIN_ROOT=$(dirname "$(dirname "$cxx")")" >> "$GITHUB_ENV" + echo "$HOME/qemu-hexagon-plugins" >> "$GITHUB_PATH" + if "$HOME/qemu-hexagon-plugins/qemu-hexagon" -plugin help 2>&1 | grep -q "unknown option"; then + echo "::error::built qemu-hexagon lacks plugin support"; exit 1 + fi + # These versions go into bench/README.md next to the numbers when a key # is seeded or re-recorded; the counts are only comparable within a # toolchain/QEMU pair. - name: Toolchain versions run: | - arm-none-eabi-gcc --version | head -1 - qemu-system-arm --version | head -1 + if [ "${{ matrix.arch }}" = hexagon ]; then + "$HEXAGON_TOOLCHAIN_ROOT/bin/hexagon-unknown-linux-musl-clang++" --version | head -1 + qemu-hexagon --version | head -1 + else + arm-none-eabi-gcc --version | head -1 + qemu-system-arm --version | head -1 + fi cmake --version | head -1 - name: Build counting plugin @@ -187,6 +263,8 @@ jobs: CEILING_Q31: ${{ matrix.text_ceiling_q31 }} run: | set -eo pipefail + SIZE_TOOL=arm-none-eabi-size + if [ "${{ matrix.arch }}" = hexagon ]; then SIZE_TOOL="$HEXAGON_TOOLCHAIN_ROOT/bin/llvm-size"; fi cmake -S . -B "$SIZE_DIR" \ -DCMAKE_BUILD_TYPE=MinSizeRel \ -DCMAKE_TOOLCHAIN_FILE=${{ matrix.toolchain }} \ @@ -198,7 +276,7 @@ jobs: profile=${entry%%:*} ceiling=${entry#*:} probe="$SIZE_DIR/bench/tap_dsp_size_probe_rfft_${profile}_512" - arm-none-eabi-size -A "$probe" | tee "size-$profile.txt" + $SIZE_TOOL -A "$probe" | tee "size-$profile.txt" text=$(awk '$1 == ".text" { print $2 }' "size-$profile.txt") case "$text" in ''|*[!0-9]*) echo "::error::$KEY: $profile probe .text unavailable (no numeric .text row in size-$profile.txt: '$text')"; exit 1 ;; @@ -289,7 +367,7 @@ jobs: # Folds the per-key seeding artifacts (each has one key filled in) into the # single file the seeding commit copies to bench/baselines.json. Fails when # any icount job failed, so a partial seed from a half-red run can never be - # merged. NEVER a required check: the five `icount ` jobs are the + # merged. NEVER a required check: the six `icount ` jobs are the # required checks once seeded (bench/README.md). seed-summary: name: Merge seeding artifacts (never a required check) diff --git a/cmake/hexagon-linux-musl.cmake b/cmake/hexagon-linux-musl.cmake new file mode 100644 index 0000000..20a92b6 --- /dev/null +++ b/cmake/hexagon-linux-musl.cmake @@ -0,0 +1,63 @@ +# SPDX-License-Identifier: MIT +# Copyright 2026 Timothy Place and the DspTap contributors. +# Adapted from MuTap's cmake/hexagon-linux-musl.cmake (MIT, MuTap contributors). +# +# Cross-compile DspTap for Qualcomm Hexagon: triple hexagon-unknown-linux-musl, +# built with the CodeLinaro "toolchain for hexagon" (clang + musl sysroot + LLVM +# runtimes). Point HEXAGON_TOOLCHAIN_ROOT (cache variable or environment) at +# the unpacked clang+llvm-*-cross-hexagon-unknown-linux-musl/x86_64-linux-gnu +# directory. Used by bench.yml's `hexagon` key (the instruction-count ratchet +# under qemu-hexagon user-mode emulation); MuTap's Hexagon CI leg and ratchet +# use the same toolchain, flags and QEMU, so the two repositories' Hexagon +# counts are comparable. +# +# A hosted Linux target, not bare metal: the bench binaries print through +# stdio and exit normally, and are linked statically (first-class with musl) +# so the emulator needs no sysroot path. +# +# Flags (MuTap's, kept identical): +# -mv68 the newest revision the shipped sysroot libraries +# and qemu-hexagon agree on. +# -mhvx -mhvx-length=128b the 128-byte HVX unit, for auto-vectorization. + +set(CMAKE_SYSTEM_NAME Linux) +set(CMAKE_SYSTEM_PROCESSOR Hexagon) + +if(NOT DEFINED HEXAGON_TOOLCHAIN_ROOT OR HEXAGON_TOOLCHAIN_ROOT STREQUAL "") + set(HEXAGON_TOOLCHAIN_ROOT "$ENV{HEXAGON_TOOLCHAIN_ROOT}") +endif() +if(HEXAGON_TOOLCHAIN_ROOT STREQUAL "") + message(FATAL_ERROR + "Set HEXAGON_TOOLCHAIN_ROOT (cache variable or environment) to the unpacked CodeLinaro " + "hexagon toolchain (clang+llvm-*-cross-hexagon-unknown-linux-musl/x86_64-linux-gnu)") +endif() +# try_compile projects re-read this file without the cache: pass the root on. +set(ENV{HEXAGON_TOOLCHAIN_ROOT} "${HEXAGON_TOOLCHAIN_ROOT}") + +set(TAP_DSP_HEXAGON_TRIPLE hexagon-unknown-linux-musl) +if(EXISTS "${HEXAGON_TOOLCHAIN_ROOT}/bin/${TAP_DSP_HEXAGON_TRIPLE}-clang++") + set(CMAKE_C_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/${TAP_DSP_HEXAGON_TRIPLE}-clang") + set(CMAKE_CXX_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/${TAP_DSP_HEXAGON_TRIPLE}-clang++") +else() + set(CMAKE_C_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/clang") + set(CMAKE_CXX_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/clang++") + set(CMAKE_C_COMPILER_TARGET ${TAP_DSP_HEXAGON_TRIPLE}) + set(CMAKE_CXX_COMPILER_TARGET ${TAP_DSP_HEXAGON_TRIPLE}) +endif() + +set(TAP_DSP_HEXAGON_ARCH_FLAGS "-mv68 -mhvx -mhvx-length=128b") +set(CMAKE_C_FLAGS_INIT "${TAP_DSP_HEXAGON_ARCH_FLAGS}") +set(CMAKE_CXX_FLAGS_INIT "${TAP_DSP_HEXAGON_ARCH_FLAGS}") +# --eh-frame-hdr: not implied for -static links, but the unwinder needs +# PT_GNU_EH_FRAME to find the exception tables. +set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -Wl,--eh-frame-hdr") + +set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER) +set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY) + +find_program(TAP_DSP_QEMU_HEXAGON NAMES qemu-hexagon-static qemu-hexagon) +if(TAP_DSP_QEMU_HEXAGON) + set(CMAKE_CROSSCOMPILING_EMULATOR "${TAP_DSP_QEMU_HEXAGON}") +endif() diff --git a/scripts/icount.py b/scripts/icount.py index 1a2e1ca..cbd1835 100755 --- a/scripts/icount.py +++ b/scripts/icount.py @@ -8,7 +8,7 @@ instruction-counting plugin (tools/qemu_insn_plugin), then compares against bench/baselines.json. - icount.py --target {m4-softfp,m4f,m33,m55,m55-ooura} --build-dir DIR + icount.py --target {m4-softfp,m4f,m33,m55,m55-ooura,hexagon} --build-dir DIR --plugin LIB [--update] [--record FILE] [--baselines bench/baselines.json] [--tolerance 0.03] icount.py --merge FILE [FILE ...] [--baselines bench/baselines.json] @@ -62,6 +62,10 @@ "m33": "mps2-an505", "m55": "mps3-an547", "m55-ooura": "mps3-an547", + # Not a system model: Hexagon runs hosted (hexagon-unknown-linux-musl, + # static) under qemu-hexagon user-mode emulation, built with plugin + # support (bench.yml; cmake/hexagon-linux-musl.cmake). + "hexagon": None, } PREFIX = "tap_dsp_icount_" DONE_MARKER = "TAP_DSP_ICOUNT_DONE ok=1" @@ -72,9 +76,11 @@ def qemu_cmd(target: str, plugin: str, binary: str) -> list[str]: # "-d plugin" routes qemu_plugin_outs() to stderr; without it the count # line is silently dropped. - machine = MACHINES.get(target) - if machine is None: + if target not in MACHINES: raise SystemExit(f"unknown target {target}") + if target == "hexagon": + return ["qemu-hexagon", "-d", "plugin", "-plugin", plugin, binary] + machine = MACHINES[target] return ["qemu-system-arm", "-M", machine, "-nographic", "-semihosting", "-d", "plugin", "-plugin", plugin, "-kernel", binary] From c79656cf16345f76f34484b5e99489d63a65f438 Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Mon, 28 Sep 2026 00:33:07 +0000 Subject: [PATCH 2/7] srdif: tune the floating engine for Hexagon (clang) without costing the Arm keys The srdif engine (tap/DspTap#42) executed more instructions on the new `hexagon` ratchet key (clang 19.1.5, -mv68 -mhvx, qemu-hexagon 8.2.2) than the engine it replaced: +15.9 % / +19.8 % (float, N = 512 / 2048) and +7.2 % (double). Now -7.64 % / -5.88 % and -3.20 % below it, and every Cortex-M float scenario falls as well (-0.06 ... -1.11 %). - Under clang (not at -Os / -Oz), the butterflies, groups, compile-time blocks, their index-sequence lambdas, the post-pass pairs and the swaps are forced inline (TAP_DSP_SRDIF_INLINE). clang had kept them out of line on Hexagon: calls per group and register leaves through a stack array. GCC keeps its own choices (forced there, the FPU keys cost 4.5 - 9.1 % more). - The post-pass runs two bin pairs with every load before any store (post_two), so the pairs' arithmetic can share Hexagon packets. - Blocks of 16 and fewer run one split-radix level at a time in memory down to pairs (level, level_bf) instead of register leaves: a leaf of 16 holds 32 values and spills on every target measured. - Double runs the run-time fused pass one group per loop step (two groups of doubles are twice Hexagon's register file). - mul_sub spells a b - c d so clang's statement-scoped contraction fuses a multiply-subtract instead of negating a product (float only). No output bit moves at -ffp-contract=off: the srdif fingerprints pass unchanged on the host (g++, clang++), the four QEMU legs and Hexagon. Contracting builds move last bits (bench checksums on the FPU keys and Hexagon float). Contract, tables and heap unchanged. The measurements and what was tried are in docs/fft-design.md, "Hexagon (clang) tuning". Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- include/tap/dsp/fft/srdif.h | 335 +++++++++++++++++++++++++----------- tests/test_fft_routing.cpp | 17 +- 2 files changed, 248 insertions(+), 104 deletions(-) diff --git a/include/tap/dsp/fft/srdif.h b/include/tap/dsp/fft/srdif.h index 887ac3d..4e6d5a8 100644 --- a/include/tap/dsp/fft/srdif.h +++ b/include/tap/dsp/fft/srdif.h @@ -66,7 +66,7 @@ // tables.h does for the fixed-point tables. // // The arrangement of these pieces (the fused two-level pass, the twiddle -// pairing, the compile-time blocks and register leaves, the table layouts) +// pairing, the compile-time blocks and their in-memory levels, the table layouts) // is derived in the docstrings below. #pragma once @@ -93,6 +93,28 @@ #define TAP_DSP_SRDIF_NOINLINE #endif +// Under clang, the kernel's building blocks (butterflies, groups, the +// compile-time blocks and their levels, the post-pass pairs, the permutation's +// swaps) are forced inline into the pass that calls them (a performance +// attribute only; the arithmetic and its order are the same either way). +// clang's inliner, left to itself, kept them out of line on Hexagon (clang +// 19, v68): the register leaves this engine had then went through a stack +// array and every group through a call. Forcing them was the largest single +// step of the Hexagon tuning (rfft_f32_512 57.66 -> 49.67 M instructions +// with the leaves of the time; the tree as it stands reads 54.13 M without +// the attribute and 45.94 M with it: docs/fft-design.md, "Hexagon (clang) +// tuning"). GCC is left to its own choices: forced there, the Cortex-M +// keys with an FPU (M4F, M33, M55 without CMSIS) cost 4.5 - 9.1 % more +// instructions (spills in the larger bodies; the soft-float M4 0.1 - 0.2 % +// less). Not forced when optimizing for size (-Os / -Oz define +// __OPTIMIZE_SIZE__): forced, the Hexagon float size probe's .text grows +// from 238,628 to 263,204 bytes. +#if defined(__clang__) && !defined(__OPTIMIZE_SIZE__) +#define TAP_DSP_SRDIF_INLINE __attribute__((always_inline)) +#else +#define TAP_DSP_SRDIF_INLINE +#endif + namespace tap::dsp::detail { /// The engine's trigonometry: cos and sin of 2 pi k / 2^L in integer @@ -339,8 +361,9 @@ namespace tap::dsp::detail { /// 48 of each. Every block of l >= 32 is processed this way and /// recurses on the five blocks the two levels leave: l/4 (the first /// half's half), l/8 twice (its quarters) and l/4 twice (the - /// quarters). Groups run two per loop step in the run-time pass - /// (l >= 128) so its eight row pointers serve both. + /// quarters). In float, groups run two per loop step in the + /// run-time pass (l >= 128) so its eight row pointers serve both; in + /// double one (k_one_group_per_step). /// - Special butterflies, as Sorensen, Heideman and Burrus count them: /// j = 0 (W = 1, additions only) and j = l/8 (W^j = (1 + i)/sqrt 2, /// W^3j = (-1 + i)/sqrt 2: two additions and two multiplications per @@ -355,9 +378,9 @@ namespace tap::dsp::detail { /// walks it backwards (group_high). /// - Compile-time blocks: blocks of 32 and 64 (block<32>, block<64>) /// run with every offset a constant and their own small tables, and - /// blocks of 16 and fewer are register leaves (leaf: loaded, - /// transformed through every remaining level and stored once). The - /// run-time recursion (kernel) handles l >= 128 only. + /// blocks of 16 and fewer run one split-radix level at a time in + /// memory (level, one butterfly at a time) down to blocks of two. + /// The run-time recursion (kernel) handles l >= 128 only. /// /// The tables: one allocation, sized once and built in place by the /// constructor, never touched by a transform except to read. In Samples, @@ -438,11 +461,16 @@ namespace tap::dsp::detail { /// Speed: the instruction-count ratchet's float scenarios (forward plus /// scaled inverse over 2^20 samples, harness included; arm-none-eabi-gcc /// 13.2.1 -O3, QEMU 8.2.2), against the engine it replaced, N = 512 / - /// 2048: Cortex-M4 soft-float -2.71 % / -2.23 %, M4F -4.70 % / -4.49 %, - /// M33 -7.89 % / -7.53 %, M55 without CMSIS -5.98 % / -5.04 % - /// (bench/README.md, docs/fft-design.md, "The floating engine (srdif)"). - /// On x86-64 (a same-machine A/B of the targets sheet's hostbench, g++ - /// 13.3, pinned core, 11 runs; the predecessor's medians are review A's + /// 2048: Cortex-M4 soft-float -2.78 % / -2.28 %, M4F -5.76 % / -5.51 %, + /// M33 -8.24 % / -7.98 %, M55 without CMSIS -6.33 % / -5.52 % (at #42, + /// before the Hexagon tuning: -2.71 / -2.23, -4.70 / -4.49, -7.89 / + /// -7.53, -5.98 / -5.04 %); Hexagon (clang 19.1.5 -O3, -mv68 -mhvx, + /// qemu-hexagon 8.2.2, which counts packets) float -7.64 % / -5.88 %, + /// double at N = 512 -3.20 %, where #42 read +15.9 % / +19.8 % and + /// +7.2 % (bench/README.md; docs/fft-design.md, "The floating engine + /// (srdif)" and its "Hexagon (clang) tuning"). On x86-64 (a same-machine + /// A/B of the targets sheet's hostbench, g++ 13.3, pinned core, 11 runs, + /// at #42, before the Hexagon tuning; the predecessor's medians are review A's /// of tap/DspTap#42, this engine's measured the same way on the same /// machine, within 2 % of review A's own re-run of the previous tree), /// N = 256 … 4096, forward / inverse against the engine it replaced: @@ -659,7 +687,7 @@ namespace tap::dsp::detail { } } - static void swap2(Sample* p, Sample* q) noexcept { + TAP_DSP_SRDIF_INLINE static void swap2(Sample* p, Sample* q) noexcept { const Sample p0 = p[0]; const Sample p1 = p[1]; const Sample q0 = q[0]; @@ -681,12 +709,39 @@ namespace tap::dsp::detail { Sample i; }; - static cv ld(const Sample* p) noexcept { return {p[0], p[1]}; } - static void st(Sample* p, cv v) noexcept { + TAP_DSP_SRDIF_INLINE static cv ld(const Sample* p) noexcept { return {p[0], p[1]}; } + TAP_DSP_SRDIF_INLINE static void st(Sample* p, cv v) noexcept { p[0] = v.r; p[1] = v.i; } + /// a b - c d, the first product rounded in a statement of its own. + /// + /// Without contraction this is the same two products and one + /// subtraction, rounded the same way, as the plain expression (the + /// output bits and the fingerprints do not move). The float spelling + /// is for the compilers that contract within a statement (clang, + /// AppleClang): on `a * b - c * d` clang fuses the left product and + /// negates the right one, fma(a, b, -(c d)), which costs Hexagon a + /// separate negation in the two slots the float arithmetic itself + /// needs; here the statement is `ab - c * d`, which it fuses as + /// fma(-c, d, ab), one Hexagon multiply-subtract (Rx -= sfmpy(Rs, + /// Rt)). Hexagon float scenarios -0.65 % / -0.43 % (N = 512 / 2048). + /// GCC contracts across statements after SSA and compiles both + /// spellings the same (the Cortex-M counts are identical). Double + /// keeps the plain expression: Hexagon has no double fused + /// multiply-add, and the split spelling there only moved the schedule + /// (rfft_f64_512 +0.14 %). + TAP_DSP_SRDIF_INLINE static Sample mul_sub(Sample a, Sample b, Sample c, Sample d) noexcept { + if constexpr (std::is_same_v) { + const Sample ab = a * b; + return ab - c * d; + } + else { + return a * b - c * d; + } + } + /// The three kinds of split-radix butterfly, by twiddle. enum class tw { one, eighth, general }; @@ -698,7 +753,7 @@ namespace tap::dsp::detail { /// eighth: j = l/8, W^j = (1 + i)/sqrt 2, W^3j = (-1 + i)/sqrt 2; /// general: w1 = W^j, w3 = W^3j as given. template - static void bf(cv& x0, cv& x1, cv& x2, cv& x3, cv w1 = {}, cv w3 = {}) noexcept { + TAP_DSP_SRDIF_INLINE static void bf(cv& x0, cv& x1, cv& x2, cv& x3, cv w1 = {}, cv w3 = {}) noexcept { const Sample t1r = x0.r - x2.r; const Sample t1i = x0.i - x2.i; const Sample t2r = x1.r - x3.r; @@ -743,12 +798,12 @@ namespace tap::dsp::detail { } else { if constexpr (Inverse) { - x2 = {ur * w1.r + ui * w1.i, ui * w1.r - ur * w1.i}; - x3 = {vr * w3.r + vi * w3.i, vi * w3.r - vr * w3.i}; + x2 = {ur * w1.r + ui * w1.i, mul_sub(ui, w1.r, ur, w1.i)}; + x3 = {vr * w3.r + vi * w3.i, mul_sub(vi, w3.r, vr, w3.i)}; } else { - x2 = {ur * w1.r - ui * w1.i, ur * w1.i + ui * w1.r}; - x3 = {vr * w3.r - vi * w3.i, vr * w3.i + vi * w3.r}; + x2 = {mul_sub(ur, w1.r, ui, w1.i), ur * w1.i + ui * w1.r}; + x3 = {mul_sub(vr, w3.r, vi, w3.i), vr * w3.i + vi * w3.r}; } } } @@ -759,7 +814,8 @@ namespace tap::dsp::detail { /// level-l/2 butterfly at j of the first half (c1, c3), whose four /// inputs are exactly the first-half outputs of the other two. template - static void group(Sample* p, std::size_t e, cv a1, cv a3, cv b1, cv b3, cv c1, cv c3) noexcept { + TAP_DSP_SRDIF_INLINE static void group(Sample* p, std::size_t e, cv a1, cv a3, cv b1, cv b3, cv c1, + cv c3) noexcept { const std::size_t d = 2 * e; cv x0 = ld(p); cv x2 = ld(p + 2 * d); @@ -784,21 +840,21 @@ namespace tap::dsp::detail { /// The four kinds of group in a fused pass over a block of length l = 8e. template - static void group_first(Sample* a, std::size_t e) noexcept { + TAP_DSP_SRDIF_INLINE static void group_first(Sample* a, std::size_t e) noexcept { // j = 0: W = 1, then the eighth-turn at j + e, then W = 1 at level l/2. group(a, e, {}, {}, {}, {}, {}, {}); } /// j in [1, e/2), twelve twiddles at w. template - static void group_low(Sample* a, std::size_t e, const Sample* w) noexcept { + TAP_DSP_SRDIF_INLINE static void group_low(Sample* a, std::size_t e, const Sample* w) noexcept { group(a, e, {w[0], w[1]}, {w[2], w[3]}, {w[4], w[5]}, {w[6], w[7]}, {w[8], w[9]}, {w[10], w[11]}); } /// j = e/2 = l/16: W_16^1, W_16^3; W_16^3, W_16^9 = -W_16^1; the eighth-turn at level l/2. template - static void group_mid(Sample* a, std::size_t e) noexcept { + TAP_DSP_SRDIF_INLINE static void group_mid(Sample* a, std::size_t e) noexcept { const cv c16{k_cos_pi8, k_sin_pi8}; const cv s16{k_sin_pi8, k_cos_pi8}; group(a, e, c16, s16, s16, {-k_cos_pi8, -k_sin_pi8}, {}, {}); @@ -808,11 +864,20 @@ namespace tap::dsp::detail { /// W^3j = -i conj W^3(j'+e), W^(j+e) = i conj W^j', W^3(j+e) = -i conj W^3j', /// and at level l/2 W^j = i conj W^j', W^3j = -i conj W^3j'. template - static void group_high(Sample* a, std::size_t e, const Sample* w) noexcept { + TAP_DSP_SRDIF_INLINE static void group_high(Sample* a, std::size_t e, const Sample* w) noexcept { group(a, e, {w[5], w[4]}, {-w[7], -w[6]}, {w[1], w[0]}, {-w[3], -w[2]}, {w[9], w[8]}, {-w[11], -w[10]}); } + /// Whether the run-time fused pass runs one group per loop step + /// (double) or two (float, whose eight row pointers then serve both + /// groups at constant offsets). Two groups of doubles hold 32 values + /// and 24 twiddles, twice Hexagon's register file, and spill; + /// measured on Hexagon, one per step: rfft_f64_512 -0.38 %, the float + /// scenarios +0.48 % / +0.99 % (N = 512 / 2048), so each keeps its + /// best. The Cortex-M keys measure float only. + static constexpr bool k_one_group_per_step = sizeof(Sample) == 8; + /// The fused pass over a block of l >= 128 complex values: level l and /// level l/2 of its first half; the table entry of j is j (m/l) steps. template @@ -821,36 +886,70 @@ namespace tap::dsp::detail { const std::size_t h = e / 2; const std::size_t stride = 12 * ((m_n / 2) / l); // Samples per table step at this length group_first(a, e); - // j = 1 alone, then two groups per step (e/2 - 1 is odd): the - // eight row pointers serve both groups at constant offsets. const Sample* w = m_tables.data() + m_n / 2 + 48 + stride - 12; - group_low(a + 2, e, w); - w += stride; - for (std::size_t j = 2; j < h; j += 2, w += 2 * stride) { - group_low(a + 2 * j, e, w); - group_low(a + 2 * j + 2, e, w + stride); + if constexpr (k_one_group_per_step) { + for (std::size_t j = 1; j < h; ++j, w += stride) { + group_low(a + 2 * j, e, w); + } + group_mid(a + 2 * h, e); + // j in (e/2, e) descending through the table: j' = e - j. + for (std::size_t j = h + 1; j < e; ++j) { + w -= stride; + group_high(a + 2 * j, e, w); + } } - group_mid(a + 2 * h, e); - // j in (e/2, e) descending through the table: j' = e - j. - w -= stride; - group_high(a + 2 * (h + 1), e, w); - for (std::size_t j = h + 2; j < e; j += 2) { - w -= 2 * stride; - group_high(a + 2 * j, e, w + stride); - group_high(a + 2 * j + 2, e, w); + else { + // j = 1 alone, then two groups per step (e/2 - 1 is odd): the + // eight row pointers serve both groups at constant offsets. + group_low(a + 2, e, w); + w += stride; + for (std::size_t j = 2; j < h; j += 2, w += 2 * stride) { + group_low(a + 2 * j, e, w); + group_low(a + 2 * j + 2, e, w + stride); + } + group_mid(a + 2 * h, e); + // j in (e/2, e) descending through the table: j' = e - j. + w -= stride; + group_high(a + 2 * (h + 1), e, w); + for (std::size_t j = h + 2; j < e; j += 2) { + w -= 2 * stride; + group_high(a + 2 * j, e, w + stride); + group_high(a + 2 * j + 2, e, w); + } } } /// A block of L <= 64 complex values with everything known at compile - /// time: the fused pass (L = 32, 64) over the small tables, or a leaf. + /// time: the fused pass over the small tables (L = 32, 64), one + /// split-radix level in memory (L = 4, 8, 16) or the final pair. + /// + /// Below 32 every level runs in memory, one butterfly at a time, + /// down to blocks of two; there are no register leaves. Until the + /// Hexagon tuning blocks of 16 and fewer were register leaves (loaded + /// once, transformed through every remaining level in registers, + /// stored once), which on the ratchet's targets costs more than it + /// saves: a leaf of 16 holds 32 values, the M4F's whole float register + /// file and all of Hexagon's general registers (a double takes two), + /// and spills. The arithmetic of every output is the same either way + /// (the output bits do not move). Measured in this tree with register + /// leaves of at most 16 / 8 / 4 values restored, against the pairs + /// (docs/fft-design.md, "Hexagon (clang) tuning"), millions of + /// instructions: Hexagon rfft_f32_512 46.10 / 45.88 / 45.89 / 45.94, + /// rfft_f32_2048 51.79 / 51.58 / 51.55 / 51.61, rfft_f64_512 101.30 / + /// 100.63 / 100.12 / 99.94; M4F rfft_f32_512 92.54 / 92.00 / 91.85 / + /// 91.55, rfft_f32_2048 106.30 / 105.83 / 105.58 / 105.15; the M33 and + /// the M55 without CMSIS 0.3 - 0.5 % below leaves of 8 as well, the + /// soft-float M4 flat. Pairs cost Hexagon float 0.1 % against its best + /// (leaves of 4 or 8) and are the best everywhere else, in one code + /// path for both profiles. template - static void block(Sample* a, const Sample* small) noexcept { + TAP_DSP_SRDIF_INLINE static void block(Sample* a, const Sample* small) noexcept { if constexpr (L >= 32) { constexpr std::size_t e = L / 8; constexpr std::size_t h = e / 2; const Sample* const w = small + (L == 32 ? 0 : 12); // this length's table group_first(a, e); - [&](std::index_sequence) { + [&](std::index_sequence) TAP_DSP_SRDIF_INLINE { (group_low(a + 2 * (J + 1), e, w + 12 * J), ...); (group_high(a + 2 * (e - 1 - J), e, w + 12 * J), ...); }(std::make_index_sequence{}); @@ -861,52 +960,57 @@ namespace tap::dsp::detail { block(a + L, small); block(a + L + L / 2, small); } + else if constexpr (L >= 4) { + level(a); + block(a, small); + if constexpr (L >= 8) { // blocks of one value are their own transforms + block(a + L, small); + block(a + L + L / 2, small); + } + } else { - (void)small; // the leaves need no table - leaf(a); + static_assert(L == 2, "blocks of 2 ... 64 values"); + (void)small; // the pairs need no table + const cv x0 = ld(a); + const cv x1 = ld(a + 2); + st(a, {x0.r + x1.r, x0.i + x1.i}); + st(a + 2, {x0.r - x1.r, x0.i - x1.i}); } } - /// The split-radix recursion over a block held in registers (a leaf of - /// 2 ... 16 complex values), block by block. + /// The first split-radix level of a block of L = 4, 8 or 16 values, in + /// memory, one butterfly at a time: j = 0 (W = 1), j = L/8 (the + /// eighth turn) and, at L = 16, j = 1 and 3 (W_16^1, W_16^3; W_16^3, + /// W_16^9 = -W_16^1). template - static void leaf_blocks(cv* v) noexcept { - if constexpr (L == 1) { - (void)v; // a block of one value is its own transform + TAP_DSP_SRDIF_INLINE static void level(Sample* a) noexcept { + [&](std::index_sequence) + TAP_DSP_SRDIF_INLINE { (level_bf(a), ...); }(std::make_index_sequence{}); + } + + template + TAP_DSP_SRDIF_INLINE static void level_bf(Sample* a) noexcept { + constexpr std::size_t q = L / 4; + cv x0 = ld(a + 2 * J); + cv x1 = ld(a + 2 * (J + q)); + cv x2 = ld(a + 2 * (J + 2 * q)); + cv x3 = ld(a + 2 * (J + 3 * q)); + if constexpr (J == 0) { + bf(x0, x1, x2, x3); } - else if constexpr (L == 2) { - const cv x0 = v[0]; - const cv x1 = v[1]; - v[0] = {x0.r + x1.r, x0.i + x1.i}; - v[1] = {x0.r - x1.r, x0.i - x1.i}; + else if constexpr (J == L / 8) { + bf(x0, x1, x2, x3); } - else if constexpr (L >= 4) { - static_assert(L <= 16, "leaves are 2 ... 16 complex values"); - constexpr std::size_t q = L / 4; - bf(v[0], v[q], v[2 * q], v[3 * q]); - if constexpr (L >= 8) { - constexpr std::size_t e = L / 8; - bf(v[e], v[q + e], v[2 * q + e], v[3 * q + e]); - } - if constexpr (L == 16) { // j = 1: W16^1, W16^3; j = 3: W16^3, W16^9 = -W16^1 - const cv c16{k_cos_pi8, k_sin_pi8}; - const cv s16{k_sin_pi8, k_cos_pi8}; - bf(v[1], v[5], v[9], v[13], c16, s16); - bf(v[3], v[7], v[11], v[15], s16, {-k_cos_pi8, -k_sin_pi8}); - } - leaf_blocks(v + L / 2); - leaf_blocks(v + 3 * q); - leaf_blocks(v); + else if constexpr (J == 1) { // L = 16: W_16^1, W_16^3 + bf(x0, x1, x2, x3, {k_cos_pi8, k_sin_pi8}, {k_sin_pi8, k_cos_pi8}); } - } - - template - static void leaf(Sample* a) noexcept { - [&](std::index_sequence) { - cv v[L] = {ld(a + 2 * I)...}; - leaf_blocks(v); - (st(a + 2 * I, v[I]), ...); - }(std::make_index_sequence{}); + else { // L = 16, J = 3: W_16^3, W_16^9 = -W_16^1 + bf(x0, x1, x2, x3, {k_sin_pi8, k_cos_pi8}, {-k_cos_pi8, -k_sin_pi8}); + } + st(a + 2 * J, x0); + st(a + 2 * (J + q), x1); + st(a + 2 * (J + 2 * q), x2); + st(a + 2 * (J + 3 * q), x3); } /// Split-radix DIF over l complex values at a (bit-reversed output). @@ -970,38 +1074,77 @@ namespace tap::dsp::detail { Sample* const end = a + m; post_pair(pk, pj, c); for (pk += 2, pj -= 2, c += 2; pk < end; pk += 4, pj -= 4, c += 4) { - post_pair(pk, pj, c); - post_pair(pk + 2, pj - 2, c + 2); + post_two(pk, pj, c); } } + /// The four outputs of one bin pair of the post-pass. + struct post_out { + Sample kr; + Sample ki; + Sample jr; + Sample ji; + }; + /// One bin pair of the post-pass: u = a[k], v = conj a[m - k], /// G = C_k (u - v) (conj C_k for the inverse), a[k] <- u - G, - /// a[m - k] <- conj(v + G). + /// a[m - k] <- conj(v + G); u, v and C_k given, the new a[k] and + /// conj a[m - k] returned. template - static void post_pair(Sample* pk, Sample* pj, const Sample* c) noexcept { - const Sample ur = pk[0]; - const Sample ui = pk[1]; - const Sample vr = pj[0]; - const Sample vi = pj[1]; + TAP_DSP_SRDIF_INLINE static post_out post_math(Sample ur, Sample ui, Sample vr, Sample vi, Sample cr, + Sample ci) noexcept { const Sample dr = ur - vr; const Sample di = ui + vi; - const Sample cr = c[0]; - const Sample ci = c[1]; Sample gr; Sample gi; if constexpr (Inverse) { gr = dr * cr + di * ci; - gi = di * cr - dr * ci; + gi = mul_sub(di, cr, dr, ci); } else { - gr = dr * cr - di * ci; + gr = mul_sub(dr, cr, di, ci); gi = dr * ci + di * cr; } - pk[0] = ur - gr; - pk[1] = ui - gi; - pj[0] = vr + gr; - pj[1] = vi - gi; + return {ur - gr, ui - gi, vr + gr, vi - gi}; + } + + template + TAP_DSP_SRDIF_INLINE static void post_pair(Sample* pk, Sample* pj, const Sample* c) noexcept { + const post_out o = post_math(pk[0], pk[1], pj[0], pj[1], c[0], c[1]); + pk[0] = o.kr; + pk[1] = o.ki; + pj[0] = o.jr; + pj[1] = o.ji; + } + + /// Two bin pairs, (k, m - k) and (k + 1, m - k - 1), every load before + /// any store. The compiler cannot tell a store through pk from a later + /// load through pj, so pair by pair the second pair's loads wait for + /// the first pair's stores, and on Hexagon, which issues up to four + /// instructions per packet and counts packets, the two pairs' arithmetic + /// then cannot share them: without this, the Hexagon scenarios read + /// +3.3 % / +3.0 % (float, N = 512 / 2048) and +1.0 % (double). The + /// Cortex-M keys pay at most 0.02 % for it (the soft-float M4). + template + TAP_DSP_SRDIF_INLINE static void post_two(Sample* pk, Sample* pj, const Sample* c) noexcept { + const Sample u0r = pk[0]; + const Sample u0i = pk[1]; + const Sample u1r = pk[2]; + const Sample u1i = pk[3]; + const Sample v1r = pj[-2]; + const Sample v1i = pj[-1]; + const Sample v0r = pj[0]; + const Sample v0i = pj[1]; + const post_out o0 = post_math(u0r, u0i, v0r, v0i, c[0], c[1]); + const post_out o1 = post_math(u1r, u1i, v1r, v1i, c[2], c[3]); + pk[0] = o0.kr; + pk[1] = o0.ki; + pk[2] = o1.kr; + pk[3] = o1.ki; + pj[-2] = o1.jr; + pj[-1] = o1.ji; + pj[0] = o0.jr; + pj[1] = o0.ji; } std::size_t m_n; diff --git a/tests/test_fft_routing.cpp b/tests/test_fft_routing.cpp index a3fe3b1..c9dc46b 100644 --- a/tests/test_fft_routing.cpp +++ b/tests/test_fft_routing.cpp @@ -47,14 +47,15 @@ namespace { // Sizes small enough for the QEMU legs (Part 10) and wide enough to cross // every code path the engine dispatches on (fft/srdif.h, "Structure"): - // the kernel of M = N/2 complex values as the register leaves alone - // (M = 2, 4, 8, 16: N = 4, 8, 16, 32), the compile-time blocks (M = 32, - // 64: N = 64, 128) and the run-time fused pass above them (M >= 128: - // N = 256 ... 4096, each fused pass recursing down to the compile-time - // blocks), and the post-pass's single-pair (N = 8) and paired-loop - // (N >= 16) forms. The certified geometries 512 and 2048 and 4096 (the - // largest the emulated legs run) are among them. An engine whose range - // excludes a size (CMSIS: 32 … 4096) is checked at the sizes it supports. + // the kernel of M = N/2 complex values as the small blocks alone (M = 2, + // 4, 8, 16: N = 4, 8, 16, 32; levels in memory down to pairs), the + // compile-time blocks (M = 32, 64: N = 64, 128) and the run-time fused + // pass above them (M >= 128: N = 256 ... 4096, each fused pass recursing + // down to the compile-time blocks), and the post-pass's single-pair + // (N = 8) and paired-loop (N >= 16) forms. The certified geometries 512 + // and 2048 and 4096 (the largest the emulated legs run) are among them. + // An engine whose range excludes a size (CMSIS: 32 … 4096) is checked at + // the sizes it supports. constexpr std::size_t k_sizes[] = {4, 8, 16, 32, 64, 128, 256, 512, 2048, 4096}; template From 21b2c8d04429b92907ede8ee002e1b51a4ad86f1 Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Mon, 28 Sep 2026 00:33:07 +0000 Subject: [PATCH 3/7] bench: re-record the Cortex-M float baselines after the srdif Hexagon tuning The srdif engine's Hexagon tuning lowers every float scenario on the four keys that run it, inside the +-3 % band (m4-softfp -0.06 / -0.06 %, m4f -1.11 / -1.06 %, m33 -0.38 / -0.49 %, m55-ooura -0.36 / -0.51 %, N = 512 / 2048); the exact counts are recorded, one row per scenario in bench/README.md. Fixed-point scenarios and the m55 (CMSIS) key are unchanged to the instruction and not re-recorded. The m4f DONE line in the README carries the new checksum (GCC fuses the unchanged operations differently; the fingerprints at -ffp-contract=off do not move). The hexagon key is seeded from CI, not here. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- bench/README.md | 14 ++++++++++++-- bench/baselines.json | 16 ++++++++-------- 2 files changed, 20 insertions(+), 10 deletions(-) diff --git a/bench/README.md b/bench/README.md index 5d1cb8c..c70d525 100644 --- a/bench/README.md +++ b/bench/README.md @@ -78,10 +78,12 @@ integer FNV-1a-64 fold over its bit pattern (`bench_common.h`), printed at the end: ``` -TAP_DSP_ICOUNT_DONE ok=1 engine=basic_real_fft backend=srdif scenario=rfft_f32_512 checksum=0xd26ebf9b45534325 +TAP_DSP_ICOUNT_DONE ok=1 engine=basic_real_fft backend=srdif scenario=rfft_f32_512 checksum=0x322a64b478d14325 ``` -(the `m4f` key's line on the srdif engine). The fold is exact and +(the `m4f` key's line on the srdif engine since its Hexagon tuning; +`0xd26ebf9b45534325` from #42 until then: the same outputs at +`-ffp-contract=off`, fused differently by GCC). The fold is exact and order-sensitive: two runs of the same binary print the same value, and a 1-ulp change in any single output changes it (verified on the port, the engine of the time, by nudging one spectrum bin at one iteration with @@ -203,6 +205,14 @@ shape is fixed so the record stays greppable: | m33 | rfft_f32_512 | 100,945,841 | 92,979,212 | −7.89 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_2048 | 102,784,096 | 97,602,563 | −5.04 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_512 | 89,276,321 | 83,933,381 | −5.98 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4-softfp | rfft_f32_2048 | 2,243,212,021 | 2,241,935,605 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4-softfp | rfft_f32_512 | 1,814,303,702 | 1,813,126,102 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4f | rfft_f32_2048 | 106,282,752 | 105,151,232 | −1.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4f | rfft_f32_512 | 92,575,609 | 91,545,465 | −1.11 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m33 | rfft_f32_2048 | 106,773,786 | 106,248,986 | −0.49 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m33 | rfft_f32_512 | 92,979,212 | 92,624,908 | −0.38 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m55-ooura | rfft_f32_2048 | 97,602,563 | 97,109,507 | −0.51 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m55-ooura | rfft_f32_512 | 83,933,381 | 83,628,229 | −0.36 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | `before` is `—` for a seed. `main SHA` is the commit on `main` whose push run measured the numbers: a pull-request head SHA stops resolving after diff --git a/bench/baselines.json b/bench/baselines.json index 662d7c5..a0c5010 100644 --- a/bench/baselines.json +++ b/bench/baselines.json @@ -1,21 +1,21 @@ { "m33": { - "rfft_f32_2048": 106773786, - "rfft_f32_512": 92979212, + "rfft_f32_2048": 106248986, + "rfft_f32_512": 92624908, "rfft_q15_512": 752189619, "rfft_q31_2048": 875801377, "rfft_q31_512": 714626257 }, "m4-softfp": { - "rfft_f32_2048": 2243212021, - "rfft_f32_512": 1814303702, + "rfft_f32_2048": 2241935605, + "rfft_f32_512": 1813126102, "rfft_q15_512": 746311273, "rfft_q31_2048": 869187566, "rfft_q31_512": 709161574 }, "m4f": { - "rfft_f32_2048": 106282752, - "rfft_f32_512": 92575609, + "rfft_f32_2048": 105151232, + "rfft_f32_512": 91545465, "rfft_q15_512": 753215637, "rfft_q31_2048": 876486909, "rfft_q31_512": 715019447 @@ -28,8 +28,8 @@ "rfft_q31_512": 657595969 }, "m55-ooura": { - "rfft_f32_2048": 97602563, - "rfft_f32_512": 83933381, + "rfft_f32_2048": 97109507, + "rfft_f32_512": 83628229, "rfft_q15_512": 684157511, "rfft_q31_2048": 806142635, "rfft_q31_512": 657595969 From be91cb681ef7decca641bfb552edbc67ad5f6978 Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Mon, 28 Sep 2026 00:34:22 +0000 Subject: [PATCH 4/7] docs: the srdif engine's Hexagon (clang) tuning A new subsection of "The floating engine (srdif)": what the hexagon key counts (qemu-hexagon counts packets, not instructions), where the count went at 72977aa, the before/after per key and scenario, each step and each change measured alone, what was tried and not kept, the output bits (the fingerprints hold at -ffp-contract=off, Hexagon included; which contracting builds' checksums move and why), the .text probes, the Hexagon test battery, host timings, and a finding for the class: on Hexagon a third of every floating scenario is basic_real_fft's out-of-place copy through musl's memcpy. The "Structure" bullet on register leaves notes what the tuning replaced; README's CMSIS ratio (1.78x -> 1.77x at N = 2048) follows the new m55-ooura count. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- README.md | 2 +- docs/fft-design.md | 202 ++++++++++++++++++++++++++++++++++++++++++++- 2 files changed, 202 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 4187ed3..001dde4 100644 --- a/README.md +++ b/README.md @@ -25,7 +25,7 @@ its size range and shareability as contract numbers: | Engine | Profiles | Selected by | Size range | Shareable | What it is | |---|---|---|---|---|---| | srdif (`fft/srdif.h`) | `double`; `float` unless a backend is on | default | 4 … 2^30 | yes | a split-radix decimation-in-frequency kernel of length N/2 (Duhamel–Hollmann 1984; Sorensen–Heideman–Burrus 1986), the bit-reversal permutation and the real post-pass of Cooley–Lewis–Welch 1970 / Sorensen et al. 1987 (the fixed-point engine's structure in floating point), written from the literature; exactly the split-radix operation count, 2N log₂N − 2N − 2 real operations per forward transform; tables from integer arithmetic (no libm), built in the constructor, so without fp-contraction the output bits are one row for every compiler and target CI runs (pinned as output fingerprints in `tests/test_fft_srdif_fingerprint.cpp`). Since tap/DspTap#42, written clean-room; it replaced, at the same contract, the C++20 port of Ooura's `rdft` (`fft/split_radix.h`) that the floating profiles ran from Stage 2b (#31) until #42 (see Provenance) | -| CMSIS-DSP Helium (`fft/backends/cmsis.h`) | `float` | `TAP_DSP_FFT_CMSIS` (default ON exactly when the compiler targets floating-point Helium, `__ARM_FEATURE_MVE & 2`: the Cortex-M55 / M85 class; OFF for every other target, other Cortex-M cores included) | **32 … 4096** (CMSIS-DSP's own init table; the constructor checks the init status since Stage 4, in a debug build — outside this range the library never initializes, and a release build is undefined behaviour with no fault promised: measured under QEMU, N = 16 and 8192 give a wrong spectrum silently, N = 4 a wrong spectrum and a corrupted heap) | no | Arm's radix-4/8 MVE real FFT, re-presented in the same contract to float epsilon; on the M55 the srdif engine (the `m55-ooura` icount key) executes 1.60× (N = 512) / 1.78× (N = 2048) the instructions of the CMSIS build (the `m55` key) over the whole ratchet scenario, transform plus the class's copy loop, 2/N scaling and checksum (`bench/README.md`) | +| CMSIS-DSP Helium (`fft/backends/cmsis.h`) | `float` | `TAP_DSP_FFT_CMSIS` (default ON exactly when the compiler targets floating-point Helium, `__ARM_FEATURE_MVE & 2`: the Cortex-M55 / M85 class; OFF for every other target, other Cortex-M cores included) | **32 … 4096** (CMSIS-DSP's own init table; the constructor checks the init status since Stage 4, in a debug build — outside this range the library never initializes, and a release build is undefined behaviour with no fault promised: measured under QEMU, N = 16 and 8192 give a wrong spectrum silently, N = 4 a wrong spectrum and a corrupted heap) | no | Arm's radix-4/8 MVE real FFT, re-presented in the same contract to float epsilon; on the M55 the srdif engine (the `m55-ooura` icount key) executes 1.60× (N = 512) / 1.77× (N = 2048) the instructions of the CMSIS build (the `m55` key) over the whole ratchet scenario, transform plus the class's copy loop, 2/N scaling and checksum (`bench/README.md`) | | Apple vDSP (`fft/backends/accelerate.h`) | `float` | `TAP_DSP_FFT_ACCELERATE` (default ON on Apple) | 4 … 2^20 | no | `vDSP_fft_zrip`, same contract to float epsilon; ~3× faster per transform on Apple Silicon is MuTap's transform-only measurement against the engine the library carried before (tap/MuTap#31), not re-measured in this repo against srdif (the same-binary parity *test* exists since Stage 4; the microbenchmark needs a Mac) | | int32 radix-4 (`fft/fixed_point.h`) | Q15, Q31 | the sample type (the second template argument is the scaling policy here) | 4 … 65536 | Q31 yes, Q15 no | one kernel over Q1.30 twiddles under two scaling policies, returning an exponent | diff --git a/docs/fft-design.md b/docs/fft-design.md index 4ed3113..66dd52f 100644 --- a/docs/fft-design.md +++ b/docs/fft-design.md @@ -1924,7 +1924,10 @@ the kernel is arranged around that: unrolled groups) from their own small tables; blocks of 16 and fewer are leaves held in registers through every remaining level. The run-time recursion handles l ≥ 128 only, and its groups run two per loop step so - the eight row pointers serve both. + the eight row pointers serve both. (As #42 shipped it. The Hexagon tuning + below replaced the register leaves with one split-radix level at a time in + memory down to pairs, and runs the double profile's run-time groups one + per step.) - **The permutation** keeps no index table: a bit-reversed counter walks beside the natural index (the reverse-carry increment of Gold and Rader, *Digital Processing of Signals*, 1969: adding one flips an index's @@ -2291,6 +2294,203 @@ made. `FloatEngineTracksDoubleAtN512`, 2.25e-7 against 9.73e-8 / 1.05e-7; `DoubleForwardTracksCompensatedDft`, 7.4e-16 against 1.49–1.70e-16). + +### Hexagon (clang) tuning + +The engine met "must not regress" on every key the ratchet measured at #42, +all four Cortex-M cores under GCC 13.2. DspTap's main consumer also ships on +Qualcomm Hexagon (clang 19, v68 with HVX-128), where the srdif engine of +`72977aa` executed more instructions than the engine it replaced; the bench +gained a `hexagon` key for it (`cmake/hexagon-linux-musl.cmake`, the +CodeLinaro clang 19.1.5 toolchain, `-mv68 -mhvx -mhvx-length=128b`, static +musl, a plugin-enabled `qemu-hexagon` 8.2.2), and the engine was tuned until +every floating Hexagon scenario is below the predecessor's count, without +costing any Cortex-M key. Done under the same clean-room rules as the +engine: from this tree, its own disassembly and counts, and the literature +the header cites. + +**What the Hexagon key counts.** `qemu-hexagon` translates a packet (up to +four instructions issued together) as one guest instruction, so the plugin +counts **packets**, not instructions: a per-address profile (a scratch +plugin, one counter per translated instruction) holds counts at +packet-start addresses only, and they sum to the ratchet's total. The +Hexagon count therefore rewards what the scheduler can pack side by side, +not only what it executes. In the disassembly the float and double +arithmetic takes two of a packet's four slots at most (with the negations +and shifts that share them), the loads and stores the other two; a double +multiply is six instructions (`dfmpyfix` twice, `dfmpyll`, `dfmpylh` twice, +`dfmpyhh`), a double add one. + +**Where the count went at `72977aa`** (`rfft_f32_512`, 57.66 M packets): +the class's out-of-place copies, which clang compiles to calls of musl's +`memcpy`, 17.0 M (29 %; 33.7 M of the double scenario's 110.7 M), and the +harness's fold 6.1 M, both the same for either engine (the fixed-point +scenarios, which do not run srdif, measure the same on both trees); the +engine the rest. clang's inliner had kept the butterflies' callers out of +line: the fused pass called `group<…>` per group, each register leaf went +through a stack array of values (`leaf_blocks(cv*)`) and a call, and the +index-sequence lambdas of `block<32 | 64>` and `leaf` were calls. GCC +inlines differently and the Cortex-M keys had not shown it. + +**Measured** with `scripts/icount.py` as `bench.yml` runs it (a fresh +Release build per key), on one machine, 2026-09-28. Hexagon counts on this +machine read a constant +1,250 against the machine that measured the +predecessor, on every scenario including the fixed-point ones, which run the +same code on both trees; the predecessor column below is that machine's +count + 1,250. The Cortex-M counts reproduce `bench/baselines.json` exactly. + +| key | scenario | `72977aa` | tuned | Δ | predecessor | tuned vs predecessor | +|---|---|---:|---:|---:|---:|---:| +| `hexagon` | `rfft_f32_512` | 57,660,946 | 45,936,146 | −20.33 % | 49,734,654 | −7.64 % | +| `hexagon` | `rfft_f32_2048` | 65,686,018 | 51,610,626 | −21.43 % | 54,836,437 | −5.88 % | +| `hexagon` | `rfft_f64_512` | 110,685,320 | 99,941,512 | −9.71 % | 103,241,260 | −3.20 % | +| `m4-softfp` | `rfft_f32_512` | 1,814,303,702 | 1,813,126,102 | −0.06 % | 1,864,929,141 | −2.78 % | +| `m4-softfp` | `rfft_f32_2048` | 2,243,212,021 | 2,241,935,605 | −0.06 % | 2,294,360,259 | −2.28 % | +| `m4f` | `rfft_f32_512` | 92,575,609 | 91,545,465 | −1.11 % | 97,138,544 | −5.76 % | +| `m4f` | `rfft_f32_2048` | 106,282,752 | 105,151,232 | −1.06 % | 111,278,416 | −5.51 % | +| `m33` | `rfft_f32_512` | 92,979,212 | 92,624,908 | −0.38 % | 100,945,841 | −8.24 % | +| `m33` | `rfft_f32_2048` | 106,773,786 | 106,248,986 | −0.49 % | 115,465,626 | −7.98 % | +| `m55-ooura` | `rfft_f32_512` | 83,933,381 | 83,628,229 | −0.36 % | 89,276,321 | −6.33 % | +| `m55-ooura` | `rfft_f32_2048` | 97,602,563 | 97,109,507 | −0.51 % | 102,784,096 | −5.52 % | + +Unchanged, to the instruction: the `m55` key (CMSIS-DSP, 52,382,329 / +54,858,158) and every fixed-point scenario on every key, Hexagon included +(159,978,609 / 157,304,960 / 184,892,304 for Q15 512 / Q31 512 / Q31 2048). +The Cortex-M float baselines are re-recorded to these counts +(`bench/README.md`); the Hexagon key is seeded from CI. Against CMSIS-DSP +on the M55 the srdif engine now executes 1.60× (N = 512) / 1.77× +(N = 2048) the instructions (1.60× / 1.78× before). + +**What paid, step by step** (Hexagon, each step on top of the one before; +`rfft_f32_512` / `rfft_f32_2048` / `rfft_f64_512`, millions of packets): + +| step | f32 512 | f32 2048 | f64 512 | +|---|---:|---:|---:| +| `72977aa` | 57.66 | 65.69 | 110.69 | +| clang only: butterflies, groups, leaves, post-pass pairs and swaps forced inline (`TAP_DSP_SRDIF_INLINE`) | 49.67 | 55.24 | 105.83 | +| post-pass: two bin pairs with every load before any store (`post_two`) | 48.26 | 53.80 | 104.54 | +| the first level of a block of 16 (and of 8, 4 in double) in memory, register leaves of 8 (float) / 2 (double) | 48.15 | 53.70 | 103.02 | +| forced inline also the index-sequence lambdas and the compile-time blocks | 46.10 | 51.72 | 100.44 | +| double: the run-time fused pass one group per loop step | 46.10 | 51.72 | 100.06 | +| `mul_sub`: `a b − c d` spelled so clang fuses it as a multiply-subtract | 45.88 | 51.58 | 100.09 | +| no register leaves at all: levels in memory down to pairs, both profiles | 45.94 | 51.61 | 100.08 | +| `mul_sub`'s split spelling for float only | 45.94 | 51.61 | 99.94 | + +Each change measured alone against the final tree (the change undone, +everything else kept; Hexagon f32 512 / f32 2048 / f64 512, then the +Cortex-M keys): + +- **Forced inlining (clang only).** Without it: 54.13 / 62.12 / 107.12 M + (+17.8 / +20.4 / +7.2 %). Forced under GCC as well: `m4f` +9.1 / +6.8 %, + `m33` +5.7 / +4.5 %, `m55-ooura` +6.3 / +4.9 % (spills in the larger + bodies), `m4-softfp` −0.22 / −0.14 %; so GCC keeps its own choices. Not + forced under `-Os` / `-Oz` (`__OPTIMIZE_SIZE__`): forced, the Hexagon + MinSizeRel float probe's `.text` grows from 238,628 to 263,204 bytes. +- **`post_two`.** Pair by pair the compiler cannot move the second pair's + loads above the first pair's stores (a store through `pk` may alias a + load through `pj` as far as it knows), so the two pairs' arithmetic + cannot share packets. Without it: 47.47 / 53.17 / 100.97 M (+3.3 / +3.0 / + +1.0 %). The Cortex-M keys pay 0.00 – 0.02 % for it. +- **Levels in memory down to pairs.** Register leaves of at most 16 / 8 / 4 + values restored in the final tree: Hexagon 46.10 / 45.88 / 45.89 (f32 512), + 51.79 / 51.58 / 51.55 (f32 2048), 101.30 / 100.63 / 100.12 (f64), against + 45.94 / 51.61 / 99.94 for pairs; `m4f` 92.54 / 92.00 / 91.85 and 106.30 / + 105.83 / 105.58 against 91.55 / 105.15; `m33` and `m55-ooura` 0.3 – 0.5 % + below leaves of 8, `m4-softfp` flat. A leaf of 16 holds 32 values: the + M4F's whole single-precision register file and all 32 of Hexagon's general + registers, of which a double takes two. Pairs cost Hexagon float 0.1 % + against its best (leaves of 4 or 8) and are the best everywhere else, in + one code path. +- **One group per step in double.** Two groups per step: 100.32 M (+0.38 %). + One per step in float too: 46.16 / 52.12 M (+0.48 / +0.99 %), so float + keeps two. +- **`mul_sub`.** clang contracts within a statement and, on `a * b − c * d`, + fuses the left product and negates the right, `fma(a, b, −(c d))`: a + separate negation (`togglebit`) in the two slots the float arithmetic + needs. `ab − c * d` with `ab` a statement of its own fuses as + `fma(−c, d, ab)`, one `Rx −= sfmpy(Rs, Rt)`. Without it: float 46.24 / + 51.83 M (+0.65 / +0.43 %). Double keeps the plain expression (Hexagon has + no double fused multiply-add; the split spelling only moved the schedule, + +0.14 %). Without contraction both spellings are the same two products + and one subtraction; GCC, which contracts after SSA across statements, + compiles both the same (identical Cortex-M counts). + +Tried and not kept: + +- All eight loads of a group before its stores (the level-l butterfly at + j + l/8 no longer waits for the stores of the one at j): +1.8 / +3.2 % + float, +0.1 % double on Hexagon at the time; the extra live values cost + more than the freedom. +- The permutation's six unconditional exchanges as two batches of three, + every load before any store (measured at the `mul_sub` step): Hexagon + float −0.39 / −0.37 %, but double +0.34 % (−0.05 % in batches of two), + `m55-ooura` +1.69 / +1.47 % and `m4-softfp` +0.10 / +0.08 %. +- Register-leaf sizes of 16, 8 and 4 (above). + +**Output bits.** Unchanged wherever the fingerprints are defined: every +change reorders loads, stores, calls and loop steps, or (`mul_sub`) splits a +statement without changing its operations, so without contraction every +output is the same operations in the same order. `tap_dsp_srdif_fingerprint` +(`-ffp-contract=off`) passes unchanged on x86-64 g++ 13.3 and clang++ 18.1, +on the four QEMU legs, and on Hexagon (clang 19.1.5 under `qemu-hexagon`, +every N from 4 to 65536, both profiles; at `72977aa` as well): **Hexagon +matches the one row**, though CI does not run it there. Under default flags, +which contract, the bench checksums move where an FMA exists: the Cortex-M +FPU keys' `rfft_f32_512` / `rfft_f32_2048` lines (`m4f`, `m33` and +`m55-ooura` alike) from `0xd26ebf9b45534325` / `0xd06aa150d5bd9325` to +`0x322a64b478d14325` / `0xb7eeda0471060d25` (GCC fuses across the former +register leaves differently once their levels run through memory; `post_two` +and `mul_sub` alone leave the GCC lines as they were), and Hexagon float +from `0x8407c06ac8ac8325` / `0xe3dffd98aa4cf725` to `0x46adbb5e3ebf2325` / +`0x7ecf6080a0f94125` (`mul_sub`). Without an FMA nothing moves: x86-64 +(`0x5caf505901a97f25`, g++ and clang++), `m4-softfp`, Hexagon double. The +error statistics are the contraction policy's business ("The fp-contraction +policy"); no accuracy re-run was needed, since no bit moved at +`-ffp-contract=off`. + +**Contract.** Unchanged: packing, sign, scale, sizes 4 … 2^30, +shareability, `noexcept` and allocation-free transforms, the tables and the +heap formula (nothing about the tables changed). The Hexagon test battery +(`tap_dsp_tests` and the side targets built with the Hexagon toolchain file, +run by ctest through `qemu-hexagon`, sweeps capped at 4096) passes 423 of +426; the three failures are `fft_abi_tag`'s `dlopen` tests, which a static +musl image cannot run ("Dynamic loading not supported"). + +**`.text`** of the MinSizeRel float probe: `m4-softfp` 31,465 B (31,081 +before), `m4f` 28,985 (28,921), `m33` 28,385 (28,305), `m55-ooura` 28,289 +(28,161), all under their ceilings, which are not re-recorded; Hexagon +238,628 (239,044); the Q15 / Q31 probes and the `m55` CMSIS probe +unchanged. + +**Host timings** (informational; `bench/bench_fft.cpp`, `-O3 -DNDEBUG`, +g++ 13.3 and clang++ 18.1, Xeon 2.8 GHz VM (4 vCPUs, load about 1), pinned to one core, 11 runs +alternating the two trees, the median of the per-run minima, ns per +transform; `72977aa` → tuned): + +| compiler | scenario | forward | inverse | +|---|---|---:|---:| +| g++ | `rfft_f32_512` | 1,714 → 1,682 (−1.9 %) | 1,788 → 1,780 (−0.5 %) | +| g++ | `rfft_f32_2048` | 8,259 → 8,022 (−2.9 %) | 8,605 → 8,398 (−2.4 %) | +| g++ | `rfft_f64_512` | 1,852 → 1,759 (−5.0 %) | 1,965 → 1,879 (−4.4 %) | +| clang++ | `rfft_f32_512` | 1,252 → 1,283 (+2.5 %) | 1,396 → 1,288 (−7.8 %) | +| clang++ | `rfft_f32_2048` | 6,136 → 6,279 (+2.3 %) | 6,753 → 6,339 (−6.1 %) | +| clang++ | `rfft_f64_512` | 1,516 → 1,404 (−7.4 %) | 1,556 → 1,449 (−6.8 %) | + +Nothing is slower by more than 5 % (the clang++ float forward, +2.3 … ++2.5 %, is the only cell that rose); faster by 5 % or more: the g++ double +forward and every clang++ double and float-inverse cell. A different +machine from #42's table above, so the two tables do not compare cell by +cell. + +**A finding for the class, not the engine.** On Hexagon a third of every +floating scenario is `basic_real_fft`'s out-of-place copy (`forward()` and +`inverse()` copy the input before transforming in place), which clang turns +into calls of musl's `memcpy`: 17.0 M of `rfft_f32_512`'s count, 16.8 M of +`rfft_f32_2048`'s, 33.7 M of `rfft_f64_512`'s, the same for any engine. +Consumers that transform in place do not pay it; the ratchet does. Left +alone here: it is outside the engine, and the fixed-point class copies the +same way. + ## Provenance and licensing ### Where the code came from From 38a95e2c64e4225aef72b6d1841d632985b32052 Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Mon, 28 Sep 2026 00:52:08 +0000 Subject: [PATCH 5/7] docs: the hexagon key and the Hexagon tuning pass's clean-room record - bench/README.md: the `hexagon` row of the key table, and why the key exists. It counts packets, not instructions: qemu-hexagon reports one guest instruction per VLIW packet. Its absolute counts carry a per-machine constant. Six required `icount ` jobs once seeded. - docs/fft-design.md, provenance: "The Hexagon tuning pass, clean-room". What the orchestrator measured of the port, and how that was passed on (numbers only). The brief's exclusions. The implementer's access statement, including the single lines of forbidden sections its broad greps printed (routine names and one-line prose, no code, reported unused). The result. - CLAUDE.md: the engine no longer has register leaves. Blocks of 16 and fewer run one level at a time in memory, with forced inlining under clang. It also names the hexagon key beside the five Cortex-M ones. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- CLAUDE.md | 7 +++-- bench/README.md | 16 +++++++++++- docs/fft-design.md | 64 +++++++++++++++++++++++++++++++++++++++++++++- 3 files changed, 83 insertions(+), 4 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index dbb4ba8..9d6e159 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -40,7 +40,8 @@ asset's contract summary. each floor a measured number against the double profile. The floating profiles run the srdif engine (`fft/srdif.h`): the N real samples as N/2 complex values, a split-radix decimation-in-frequency kernel of length N/2 (fused two-level passes, compile-time blocks of 32 - and 64, register leaves of 16 and fewer), the bit-reversal permutation and the real post-pass + and 64, one level at a time in memory below that, forced inlining under clang for Hexagon), the + bit-reversal permutation and the real post-pass derived for the fixed-point engine (#39), written clean-room from the literature the header cites (tap/DspTap#42; it replaced, at the same contract, the C++20 port of Ooura's `rdft`, `fft/split_radix.h`, that the floating profiles ran from Stage 2b, so DspTap ships no code derived @@ -102,7 +103,9 @@ parity suite runs for real. Each leg builds `tap_dsp_tests` one-shot (`tests/bar with the `MAIN_FILTER` in `tests/CMakeLists.txt` naming what the emulation budget excludes, plus the srdif fingerprints, with every FFT sweep capped at `TAP_DSP_TEST_MAX_FFT_N=4096`. The instruction-count ratchet (`bench.yml`, `bench/README.md`) runs the same four cores under five -baseline keys at ±3 %, with `.text` ceilings per profile; a red ratchet is a failing check. The +baseline keys at ±3 %, plus a `hexagon` key (clang 19, v68 + HVX, under `qemu-hexagon`; it +counts packets, not instructions), with `.text` ceilings per profile; a red ratchet is a failing +check. The hosted build and battery need no Arm toolchain; where `gcc-arm-none-eabi` and `qemu-system-arm` are installed, `cmake -S . -B build-m33 -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m33-mps2.cmake` reproduces a leg. diff --git a/bench/README.md b/bench/README.md index 5672b40..56705ce 100644 --- a/bench/README.md +++ b/bench/README.md @@ -147,6 +147,20 @@ audit Part 3); the outcome is in the table below. | `m33` | Cortex-M33, no MVE | `mps2-an505` | srdif engine | | `m55` | Cortex-M55, Helium | `mps3-an547` | CMSIS-DSP — the deployed profile | | `m55-ooura` | Cortex-M55, `-DTAP_DSP_FFT_CMSIS=OFF` | `mps3-an547` | srdif engine — the fallback (the key keeps the historical name of the vendored C the ratchet was seeded on) | +| `hexagon` | Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl | `qemu-hexagon` user mode (8.2.2, built with plugins) | srdif engine; hosted, so `rfft_f64_512` is built and gated too | + +The `hexagon` key counts **packets**, not instructions: qemu-hexagon reports +one guest instruction per VLIW packet (up to four instructions), as the +per-address profile of the Hexagon tuning pass showed (`docs/fft-design.md`, +"Hexagon (clang) tuning"). So on that key the ratchet rewards what the +compiler packs side by side as well as what it executes. The key was added +when MuTap's Hexagon ratchet found the srdif engine 12–13 % above the port on +its chain workloads, which no key here had measured. Its counts are +comparable with MuTap's Hexagon ratchet: same toolchain, flags and QEMU. +Absolute counts move by a few hundred to about a thousand between machines, +from the user-mode process environment. The move is constant across +scenarios on one machine (the CI runner reads 473 below one local machine), +so compare within a run. JSON carries no comments, so the provenance of every recorded set lives here, in the table below: the `main` run that recorded it, and the GCC and QEMU @@ -332,7 +346,7 @@ python3 scripts/icount.py --merge a.json b.json # fold per-key files into one regression hide in the slack — the winning commit re-records. - **The ratchet runs on every pull request and on every push to `main`** on every QEMU leg (one run per ref at a time). A red ratchet is a failing - check, not a warning. Once seeded, the five `icount ` jobs are the + check, not a warning. Once seeded, the six `icount ` jobs are the required checks; the artifact-merge job never is. - **`--update` is a written commit on its own**: the measured before/after per key and the reason go into the table above, in its fixed shape. An expected diff --git a/docs/fft-design.md b/docs/fft-design.md index 16a0bef..4c97bbb 100644 --- a/docs/fft-design.md +++ b/docs/fft-design.md @@ -942,7 +942,7 @@ host. The port and its header were deleted at tap/DspTap#42 (hand-off commit `17db855`), and the srdif engine that replaced it is not a transliteration of anything: its own constraints (statement order under D9, the integer -table generator, the register leaves) are stated in `fft/srdif.h`. The rules +table generator, the leaf blocks) are stated in `fft/srdif.h`. The rules below applied to the port from Stage 2a (tap/DspTap#28) to #42. They lived in the engine header (`include/tap/dsp/fft/split_radix.h`) so the next person would not undo them, and are recorded here because they were what the bit @@ -2841,6 +2841,68 @@ records as arithmetically the package's. The access statement is the implementer's own report. The maintainer's judgement (`NOTICE.md`): since #42 DspTap ships no code derived from the package. Not legal advice. +### The Hexagon tuning pass, clean-room (2026-09-28) + +After #42 merged, MuTap's Hexagon ratchet (clang 19, v68 + HVX) measured the +srdif engine 12–13 % above the port on its chain workloads. No DspTap key +measured Hexagon, so "must not regress" had never been checked there. The +maintainer chose to tune srdif before the consumer took it. The procedure +followed #42's: + +1. **What the orchestrator did, which had read the port.** It added the + `hexagon` bench key (toolchain file, `icount.py` target, `bench.yml` job). + It measured the port's Hexagon counts once, from a detached checkout of + `7a58ebe` built with the same harness, then deleted that checkout. It also + deleted every port build product it had made while diagnosing on Hexagon: + a public-API microbenchmark's binary and its disassembly. The implementer + received those counts as numbers only, per scenario, plus a per-transform + figure from the same microbenchmark. +2. **The brief.** The implementer was barred from: + - every copy of the port and the package, and git history before `17db855`; + - the main DspTap clone, the MuTap and MuTap-Max trees (their submodules + carry the port), the other worktrees, and the orchestrator's scratch + except the brief; + - `NOTICE.md` and the audit doc; + - every section of this note except the contract, fixed-point, + fp-contraction, Stage 4, size/count and srdif sections; + - every tap/DspTap and tap/MuTap pull request and review, and every other + FFT library's source. + + The brief listed hypotheses to measure, all about the compiler, not about + the port: clang ignoring srdif's GCC pragma, inlining boundaries, index + arithmetic, the complex-multiply form, and the cost of the permutation. +3. **Access statement (the implementer's report).** It opened nothing on the + list. The one exception: broad `grep`s over this file printed single lines + from sections it could not read. These were: + - every heading; + - three history lines; + - line 971 of the transliteration-rules history, which names the port's + routines (`makect`, the `bitrv2*` family, `cftfsub` / `cftbsub`, the + `cft*` leaves); + - a line of the bit-identity record naming `bitrv2` / `bitrv2conj` and the + port's 512-point leaf; + - lines of the provenance and library-comparison sections describing + srdif's own routines and a 16-point leaf's operation count. + + These were names and one-line prose, not code. It reported using none of + them. Its history operations were a `git show --stat` of `28befbd` and a + `git archive` of it for the baseline builds. +4. **Result.** "Hexagon (clang) tuning" in "The floating engine (srdif)" + records the changes, what was tried and the counts: + - forced inlining under clang; + - post-pass pairs loaded before either is stored; + - blocks of 16 and fewer run one level at a time in memory; + - one fused-pass group per loop step for double; + - one multiply-subtract form for float. + + Every Hexagon floating scenario ends 3–8 % below the port, and every + Cortex-M float key 0.06–1.1 % below #42's baselines. No output bit moved + at `-ffp-contract=off`. + + The implementer found that the `hexagon` key counts packets (up to four + instructions each), not instructions, so it rewards packing as well as + instruction count. + ### Comparison with other FFT libraries (2026-09-27) Review B compared the srdif engine with Ooura's package. A separate From 2bd44bb6268dcf20a0cf6a222981b5455ed9597a Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Mon, 28 Sep 2026 00:55:47 +0000 Subject: [PATCH 6/7] bench: seed the hexagon key from the workflow_dispatch run on the tuned engine Baselines: the counts of workflow_dispatch run 36363710480 on this branch (2a5fc69, seed mode), the tuned srdif engine. - rfft_f32_512: 45,934,423 - rfft_f32_2048: 51,608,903 - rfft_f64_512: 99,939,789 - Q15 / Q31: 159,976,886 / 184,890,581 / 157,303,237 For the record, the engine as #42 shipped it read 57,659,223 / 65,684,295 / 110,683,597 on the first seeding run (36357132506). The engine it replaced read 49,732,931 / 54,834,714 / 103,239,537 (local, the same harness, converted to the runner's constant). .text ceilings for the key: measured + 3 %, rounded up to 64 bytes, from llvm-size -A on the statically linked probes (musl and libc++ included): 245,824 / 251,968 / 251,456. The eight Cortex-M float rows re-recorded by the tuning pass are confirmed to the instruction by the same run (compare mode, +0.00 %). Their main SHA cells, like the hexagon rows', stay pending until the squash. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- .github/workflows/bench.yml | 6 +++--- bench/README.md | 27 ++++++++++++++++++--------- bench/baselines.json | 8 ++++++++ 3 files changed, 29 insertions(+), 12 deletions(-) diff --git a/.github/workflows/bench.yml b/.github/workflows/bench.yml index 349cbe0..147f701 100644 --- a/.github/workflows/bench.yml +++ b/.github/workflows/bench.yml @@ -128,9 +128,9 @@ jobs: arch: hexagon toolchain: cmake/hexagon-linux-musl.cmake flags: "" - text_ceiling_f32: 0 - text_ceiling_q15: 0 - text_ceiling_q31: 0 + text_ceiling_f32: 245824 + text_ceiling_q15: 251968 + text_ceiling_q31: 251456 env: # Commit the tag pointed at when pinned (tags are movable; commit SHAs # are not); the header's digest is verified on download. This SHA is diff --git a/bench/README.md b/bench/README.md index 56705ce..7189484 100644 --- a/bench/README.md +++ b/bench/README.md @@ -219,14 +219,20 @@ shape is fixed so the record stays greppable: | m33 | rfft_f32_512 | 100,945,841 | 92,979,212 | −7.89 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; confirmed to the instruction by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) (compare mode, +0.00 %) | `72977aa` (the #42 squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_2048 | 102,784,096 | 97,602,563 | −5.04 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; confirmed to the instruction by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) (compare mode, +0.00 %) | `72977aa` (the #42 squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_512 | 89,276,321 | 83,933,381 | −5.98 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; confirmed to the instruction by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) (compare mode, +0.00 %) | `72977aa` (the #42 squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m4-softfp | rfft_f32_2048 | 2,243,212,021 | 2,241,935,605 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m4-softfp | rfft_f32_512 | 1,814,303,702 | 1,813,126,102 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m4f | rfft_f32_2048 | 106,282,752 | 105,151,232 | −1.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m4f | rfft_f32_512 | 92,575,609 | 91,545,465 | −1.11 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m33 | rfft_f32_2048 | 106,773,786 | 106,248,986 | −0.49 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m33 | rfft_f32_512 | 92,979,212 | 92,624,908 | −0.38 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m55-ooura | rfft_f32_2048 | 97,602,563 | 97,109,507 | −0.51 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| m55-ooura | rfft_f32_512 | 83,933,381 | 83,628,229 | −0.36 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; **pending** the confirming `workflow_dispatch` on the branch | **pending** | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4-softfp | rfft_f32_2048 | 2,243,212,021 | 2,241,935,605 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4-softfp | rfft_f32_512 | 1,814,303,702 | 1,813,126,102 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4f | rfft_f32_2048 | 106,282,752 | 105,151,232 | −1.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4f | rfft_f32_512 | 92,575,609 | 91,545,465 | −1.11 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m33 | rfft_f32_2048 | 106,773,786 | 106,248,986 | −0.49 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m33 | rfft_f32_512 | 92,979,212 | 92,624,908 | −0.38 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m55-ooura | rfft_f32_2048 | 97,602,563 | 97,109,507 | −0.51 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m55-ooura | rfft_f32_512 | 83,933,381 | 83,628,229 | −0.36 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| hexagon | rfft_f32_2048 | — | 51,608,903 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 65,684,295 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 54,834,714 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_f32_512 | — | 45,934,423 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 57,659,223 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 49,732,931 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_f64_512 | — | 99,939,789 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 110,683,597 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 103,239,537 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q15_512 | — | 159,976,886 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 159,976,886 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 159,976,914 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q31_2048 | — | 184,890,581 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 184,890,581 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 184,890,609 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q31_512 | — | 157,303,237 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 157,303,237 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 157,303,265 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | `before` is `—` for a seed. `main SHA` is the commit on `main` whose push run measured the numbers: a pull-request head SHA stops resolving after @@ -257,7 +263,7 @@ is the seeded baseline): One row per (key, probe) each time a ceiling is recorded or re-recorded; the policy is under "Size runs in the same job" below. `.text` is the row of -`arm-none-eabi-size -A` on the MinSizeRel probe; the ceiling is what +`arm-none-eabi-size -A` (the toolchain's `llvm-size -A` on the `hexagon` key) on the MinSizeRel probe; the ceiling is what `bench.yml` carries for that key and profile. | key | probe | `.text` (bytes) | ceiling | reason | run (URL) | arm-none-eabi-gcc | @@ -281,6 +287,9 @@ policy is under "Size runs in the same job" below. `.text` is the row of | m4f | `rfft_f32_512` | 28,921 | 29,824 | **re-recorded at the srdif replacement** (tap/DspTap#42) (ceiling 45,952 before): measured + 3 %, rounded up to 64 bytes; the srdif engine builds its tables from integer arithmetic, so the float probe no longer links libm's double `sin` / `cos`; the figure is the fix pass's for review A of #42 (no swap-list builder, one table allocation), 1,112 … 1,136 B below the first srdif record | measured locally as `bench.yml` builds the probe (MinSizeRel, `size -A` `.text`), 2026-09-27; confirmed to the byte by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) on `72977aa` | 13.2.1 (15:13.2.rel1-2) | | m33 | `rfft_f32_512` | 28,305 | 29,184 | **re-recorded at the srdif replacement** (tap/DspTap#42) (ceiling 45,376 before): measured + 3 %, rounded up to 64 bytes; the srdif engine builds its tables from integer arithmetic, so the float probe no longer links libm's double `sin` / `cos`; the figure is the fix pass's for review A of #42 (no swap-list builder, one table allocation), 1,112 … 1,136 B below the first srdif record | measured locally as `bench.yml` builds the probe (MinSizeRel, `size -A` `.text`), 2026-09-27; confirmed to the byte by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) on `72977aa` | 13.2.1 (15:13.2.rel1-2) | | m55-ooura | `rfft_f32_512` | 28,161 | 29,056 | **re-recorded at the srdif replacement** (tap/DspTap#42) (ceiling 40,512 before): measured + 3 %, rounded up to 64 bytes; the srdif engine builds its tables from integer arithmetic, so the float probe no longer links libm's double `sin` / `cos`; the figure is the fix pass's for review A of #42 (no swap-list builder, one table allocation), 1,112 … 1,136 B below the first srdif record | measured locally as `bench.yml` builds the probe (MinSizeRel, `size -A` `.text`), 2026-09-27; confirmed to the byte by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) on `72977aa` | 13.2.1 (15:13.2.rel1-2) | +| hexagon | `rfft_f32_512` | 238,628 | 245,824 | **recorded for the new key**: measured + 3 %, rounded up to 64 bytes; `llvm-size -A` of the statically linked probe, so the figure includes musl libc and libc++ (most of it), not DspTap alone | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) (`2a5fc69`; MinSizeRel) | clang 19.1.5 (CodeLinaro) | +| hexagon | `rfft_q15_512` | 244,580 | 251,968 | **recorded for the new key**: measured + 3 %, rounded up to 64 bytes; `llvm-size -A` of the statically linked probe, so the figure includes musl libc and libc++ (most of it), not DspTap alone | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) (`2a5fc69`; MinSizeRel) | clang 19.1.5 (CodeLinaro) | +| hexagon | `rfft_q31_512` | 244,132 | 251,456 | **recorded for the new key**: measured + 3 %, rounded up to 64 bytes; `llvm-size -A` of the statically linked probe, so the figure includes musl libc and libc++ (most of it), not DspTap alone | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) (`2a5fc69`; MinSizeRel) | clang 19.1.5 (CodeLinaro) | ### Seeding, and how the job decides what to do diff --git a/bench/baselines.json b/bench/baselines.json index a0c5010..bb20fce 100644 --- a/bench/baselines.json +++ b/bench/baselines.json @@ -1,4 +1,12 @@ { + "hexagon": { + "rfft_f32_2048": 51608903, + "rfft_f32_512": 45934423, + "rfft_f64_512": 99939789, + "rfft_q15_512": 159976886, + "rfft_q31_2048": 184890581, + "rfft_q31_512": 157303237 + }, "m33": { "rfft_f32_2048": 106248986, "rfft_f32_512": 92624908, From 6c3b9a380be4e4347d1a16a97bd8b02615af8139 Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Mon, 28 Sep 2026 01:28:54 +0000 Subject: [PATCH 7/7] Review fix pass for #44: pin the Hexagon compiler, and correct the records bench.yml: - The CodeLinaro toolchain tarball is digest-verified (HEXAGON_TOOLCHAIN_SHA256, sha256sum -c before extraction), and its version is asserted, like the QEMU source and plugin header. - qemu-hexagon's plugin probe has its own step: a missing binary fails there, not later as an icount.py traceback. The version step sets pipefail. Records: - docs/fft-design.md, the clean-room record: everything that crossed to the implementer, verbatim. That is the port's scenario counts, the per-pair microbenchmark figures for float and double, and the N = 64 comparison. It also names the #42 harness allowance, and notes the brief's wrong "instructions, not packets". - The Hexagon offset belongs to the process (binary path and environment), not the machine: +731 for a longer path, -1,584 under env -i, and +473 to +1,723 observed against CI. The predecessor cells read +-28. The README's predecessor fixed-point cells now equal #42's (the same code). - "The Cortex-M counts reproduce exactly" is for the float rows. The fixed-point rows carry the pre-existing in-band drift from #32. - srdif.h comments only (no code change; fingerprints unchanged): - The forced-inline gate is every clang. The review's clang-18 Cortex-M33 measurement (-7.1 / -5.9 %) is cited. - The gate is per translation unit. - mul_sub's bit identity needs FLT_EVAL_METHOD == 0. - Hexagon figures are packets. - CLAUDE.md: "every clang, tuned on Hexagon". Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- .github/workflows/bench.yml | 29 +++++++++++++++++++------ CLAUDE.md | 6 +++--- bench/README.md | 26 ++++++++++++++--------- docs/fft-design.md | 39 ++++++++++++++++++++++++++-------- include/tap/dsp/fft/srdif.h | 42 ++++++++++++++++++++++++------------- 5 files changed, 100 insertions(+), 42 deletions(-) diff --git a/.github/workflows/bench.yml b/.github/workflows/bench.yml index 147f701..ca8ce60 100644 --- a/.github/workflows/bench.yml +++ b/.github/workflows/bench.yml @@ -147,6 +147,8 @@ jobs: QEMU_SRC_URL: https://download.qemu.org/qemu-8.2.2.tar.xz QEMU_SRC_SHA256: 847346c1b82c1a54b2c38f6edbd85549edeb17430b7d4d3da12620e2962bc4f3 HEXAGON_TOOLCHAIN: clang+llvm-19.1.5-cross-hexagon-unknown-linux-musl + HEXAGON_TOOLCHAIN_SHA256: 55b41922318f6331590ab7baa7f5dbdd99c109327a9c44a52c5e9878fab148c1 + HEXAGON_TOOLCHAIN_VERSION: clang version 19.1.5 PLUGIN: /tmp/libinsncount.so BUILD_DIR: build-${{ matrix.key }} SIZE_DIR: build-${{ matrix.key }}-minsizerel @@ -194,26 +196,41 @@ jobs: mkdir -p ~/qemu-hexagon-plugins cp build/qemu-hexagon ~/qemu-hexagon-plugins/ + # The compiler determines every hexagon count, so it is pinned like the + # QEMU source: digest-verified on download, and its version asserted. - name: Download the Hexagon toolchain if: matrix.arch == 'hexagon' run: | + set -eo pipefail + curl -fsSLo /tmp/hexagon-toolchain.tar.zst \ + "https://artifacts.codelinaro.org/artifactory/codelinaro-toolchain-for-hexagon/19.1.5/${HEXAGON_TOOLCHAIN}.tar.zst" + echo "$HEXAGON_TOOLCHAIN_SHA256 /tmp/hexagon-toolchain.tar.zst" | sha256sum -c - mkdir -p "$RUNNER_TEMP/hexagon-toolchain" - curl -fsSL "https://artifacts.codelinaro.org/artifactory/codelinaro-toolchain-for-hexagon/19.1.5/${HEXAGON_TOOLCHAIN}.tar.zst" \ - | tar --zstd -x -C "$RUNNER_TEMP/hexagon-toolchain" + tar --zstd -x -C "$RUNNER_TEMP/hexagon-toolchain" -f /tmp/hexagon-toolchain.tar.zst + rm /tmp/hexagon-toolchain.tar.zst cxx=$(find "$RUNNER_TEMP/hexagon-toolchain" -maxdepth 5 \( -type f -o -type l \) \ -name 'hexagon-unknown-linux-musl-clang++' | sort | head -1) test -n "$cxx" || { echo "::error::no hexagon clang++ in the toolchain archive"; exit 1; } + "$cxx" --version | head -1 | grep -qF "$HEXAGON_TOOLCHAIN_VERSION" \ + || { echo "::error::the toolchain is not $HEXAGON_TOOLCHAIN_VERSION"; exit 1; } echo "HEXAGON_TOOLCHAIN_ROOT=$(dirname "$(dirname "$cxx")")" >> "$GITHUB_ENV" echo "$HOME/qemu-hexagon-plugins" >> "$GITHUB_PATH" - if "$HOME/qemu-hexagon-plugins/qemu-hexagon" -plugin help 2>&1 | grep -q "unknown option"; then + + # linux-user qemu prints "unknown option 'plugin'" when built without + # plugin support; a missing binary is caught first. + - name: Check qemu-hexagon has plugin support + if: matrix.arch == 'hexagon' + run: | + set -o pipefail + q="$HOME/qemu-hexagon-plugins/qemu-hexagon" + test -x "$q" || { echo "::error::no qemu-hexagon at $q"; exit 1; } + if "$q" -plugin help 2>&1 | grep -q "unknown option"; then echo "::error::built qemu-hexagon lacks plugin support"; exit 1 fi - # These versions go into bench/README.md next to the numbers when a key - # is seeded or re-recorded; the counts are only comparable within a - # toolchain/QEMU pair. - name: Toolchain versions run: | + set -o pipefail if [ "${{ matrix.arch }}" = hexagon ]; then "$HEXAGON_TOOLCHAIN_ROOT/bin/hexagon-unknown-linux-musl-clang++" --version | head -1 qemu-hexagon --version | head -1 diff --git a/CLAUDE.md b/CLAUDE.md index 9d6e159..8d508d2 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -40,9 +40,9 @@ asset's contract summary. each floor a measured number against the double profile. The floating profiles run the srdif engine (`fft/srdif.h`): the N real samples as N/2 complex values, a split-radix decimation-in-frequency kernel of length N/2 (fused two-level passes, compile-time blocks of 32 - and 64, one level at a time in memory below that, forced inlining under clang for Hexagon), the - bit-reversal permutation and the real post-pass - derived for the fixed-point engine (#39), written clean-room from the literature the header cites + and 64, one level at a time in memory below that, forced inlining under every clang, tuned on + Hexagon), the bit-reversal permutation and the real post-pass derived for the fixed-point engine + (#39), written clean-room from the literature the header cites (tap/DspTap#42; it replaced, at the same contract, the C++20 port of Ooura's `rdft`, `fft/split_radix.h`, that the floating profiles ran from Stage 2b, so DspTap ships no code derived from Ooura's package — the maintainer's judgement, `NOTICE.md`). Since Stage 4 the engine is a diff --git a/bench/README.md b/bench/README.md index 7189484..c5d38f4 100644 --- a/bench/README.md +++ b/bench/README.md @@ -157,10 +157,16 @@ compiler packs side by side as well as what it executes. The key was added when MuTap's Hexagon ratchet found the srdif engine 12–13 % above the port on its chain workloads, which no key here had measured. Its counts are comparable with MuTap's Hexagon ratchet: same toolchain, flags and QEMU. -Absolute counts move by a few hundred to about a thousand between machines, -from the user-mode process environment. The move is constant across -scenarios on one machine (the CI runner reads 473 below one local machine), -so compare within a run. +Absolute counts carry a per-process offset: user-mode qemu hands the guest +its environment and the binary's path, and the count moves with them. A +44-character longer path read +731 packets, and `env -i` read −1,584. The +offset is constant across the scenarios of one setup. Local setups have read ++473 to +1,723 against the runner, and CI runs agree with each other to the +packet. Compare within a run, and take baselines from CI only (as for every +key). The predecessor cells in the `hexagon` rows below are +local measurements converted to the runner's offset, ±28 packets: a rebuild +of `7a58ebe` at another path read its float scenarios 28 below them and its +fixed-point scenarios equal to #42's. JSON carries no comments, so the provenance of every recorded set lives here, in the table below: the `main` run that recorded it, and the GCC and QEMU @@ -227,12 +233,12 @@ shape is fixed so the record stays greppable: | m33 | rfft_f32_512 | 92,979,212 | 92,624,908 | −0.38 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_2048 | 97,602,563 | 97,109,507 | −0.51 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_512 | 83,933,381 | 83,628,229 | −0.36 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | -| hexagon | rfft_f32_2048 | — | 51,608,903 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 65,684,295 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 54,834,714 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | -| hexagon | rfft_f32_512 | — | 45,934,423 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 57,659,223 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 49,732,931 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | -| hexagon | rfft_f64_512 | — | 99,939,789 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 110,683,597 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 103,239,537 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | -| hexagon | rfft_q15_512 | — | 159,976,886 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 159,976,886 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 159,976,914 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | -| hexagon | rfft_q31_2048 | — | 184,890,581 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 184,890,581 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 184,890,609 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | -| hexagon | rfft_q31_512 | — | 157,303,237 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 157,303,237 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 157,303,265 (local, the same harness, CI-equivalent). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_f32_2048 | — | 51,608,903 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 65,684,295 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 54,834,714 (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_f32_512 | — | 45,934,423 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 57,659,223 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 49,732,931 (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_f64_512 | — | 99,939,789 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 110,683,597 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 103,239,537 (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q15_512 | — | 159,976,886 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 159,976,886 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 159,976,886, the same code (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q31_2048 | — | 184,890,581 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 184,890,581 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 184,890,581, the same code (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q31_512 | — | 157,303,237 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 157,303,237 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 157,303,237, the same code (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | `before` is `—` for a seed. `main SHA` is the commit on `main` whose push run measured the numbers: a pull-request head SHA stops resolving after diff --git a/docs/fft-design.md b/docs/fft-design.md index 4c97bbb..732388e 100644 --- a/docs/fft-design.md +++ b/docs/fft-design.md @@ -2335,11 +2335,20 @@ index-sequence lambdas of `block<32 | 64>` and `leaf` were calls. GCC inlines differently and the Cortex-M keys had not shown it. **Measured** with `scripts/icount.py` as `bench.yml` runs it (a fresh -Release build per key), on one machine, 2026-09-28. Hexagon counts on this -machine read a constant +1,250 against the machine that measured the -predecessor, on every scenario including the fixed-point ones, which run the -same code on both trees; the predecessor column below is that machine's -count + 1,250. The Cortex-M counts reproduce `bench/baselines.json` exactly. +Release build per key), on one machine, 2026-09-28. The Hexagon counts carry +a per-process offset: user-mode qemu passes the guest its environment and the +binary's path, and the count moves with them (the review of the tuning +measured +731 packets for a 44-character longer path, and −1,584 under +`env -i`). The offset is constant across the scenarios of one setup. Here it +read +1,250 against the setup that measured the predecessor (the fixed-point +scenarios, which do not run srdif, read the same up to that offset), so the +predecessor column below is that setup's count + 1,250. The review rebuilt +`7a58ebe` at its own path and read the predecessor's float scenarios 28 +packets below these cells, so read them as ±28. No percentage moves. +Against CI the offsets were +1,723 (this setup) and +1,596 (the review's). +The Cortex-M float counts reproduce `bench/baselines.json` exactly. The +fixed-point rows read the same +0.49 … +1.15 % in-band drift as on `main`: +they were seeded at #32 and have not been re-recorded since. | key | scenario | `72977aa` | tuned | Δ | predecessor | tuned vs predecessor | |---|---|---:|---:|---:|---:|---:| @@ -2854,14 +2863,26 @@ followed #42's: It measured the port's Hexagon counts once, from a detached checkout of `7a58ebe` built with the same harness, then deleted that checkout. It also deleted every port build product it had made while diagnosing on Hexagon: - a public-API microbenchmark's binary and its disassembly. The implementer - received those counts as numbers only, per scenario, plus a per-transform - figure from the same microbenchmark. + a public-API microbenchmark's binary and its disassembly. What crossed to + the implementer, all numbers, verbatim from the brief: + - the port's scenario counts: 49,733,404 / 54,835,187 / 103,240,010 for + `rfft_f32_512` / `rfft_f32_2048` / `rfft_f64_512`; + - "Per transform pair (forward + inverse) at N = 512 on Hexagon float, a + public-API microbenchmark read about 14.4 k instructions for the + removed engine against 18.3 k for srdif; double about 31.4 k against + 35.0 k"; + - "at N = 64 srdif was already 1–2 % below", a second measurement of the + port, at another size, that locates the gap. + + The brief also stated that the ratchet counts instructions, not packets. + That was wrong, and the implementer showed it (item 4). 2. **The brief.** The implementer was barred from: - every copy of the port and the package, and git history before `17db855`; - the main DspTap clone, the MuTap and MuTap-Max trees (their submodules carry the port), the other worktrees, and the orchestrator's scratch - except the brief; + except the brief, the conventions file, and #42's accuracy harness and + targets sheet (`clean-engine/TARGETS.md`, `clean-engine/targets/**`, + which use only the public API and were allowed at #42 too); - `NOTICE.md` and the audit doc; - every section of this note except the contract, fixed-point, fp-contraction, Stage 4, size/count and srdif sections; diff --git a/include/tap/dsp/fft/srdif.h b/include/tap/dsp/fft/srdif.h index 4e6d5a8..c8905ea 100644 --- a/include/tap/dsp/fft/srdif.h +++ b/include/tap/dsp/fft/srdif.h @@ -100,15 +100,25 @@ // clang's inliner, left to itself, kept them out of line on Hexagon (clang // 19, v68): the register leaves this engine had then went through a stack // array and every group through a call. Forcing them was the largest single -// step of the Hexagon tuning (rfft_f32_512 57.66 -> 49.67 M instructions -// with the leaves of the time; the tree as it stands reads 54.13 M without -// the attribute and 45.94 M with it: docs/fft-design.md, "Hexagon (clang) -// tuning"). GCC is left to its own choices: forced there, the Cortex-M -// keys with an FPU (M4F, M33, M55 without CMSIS) cost 4.5 - 9.1 % more -// instructions (spills in the larger bodies; the soft-float M4 0.1 - 0.2 % -// less). Not forced when optimizing for size (-Os / -Oz define -// __OPTIMIZE_SIZE__): forced, the Hexagon float size probe's .text grows -// from 238,628 to 263,204 bytes. +// step of the Hexagon tuning (rfft_f32_512 57.66 -> 49.67 M packets, the +// unit the hexagon key counts, with the leaves of the time; the tree as it +// stands reads 54.13 M without the attribute and 45.94 M with it: +// docs/fft-design.md, "Hexagon (clang) tuning"). The gate is every clang, +// not Hexagon alone (AppleClang, clang-cl, IntelLLVM and clang-based Arm +// toolchains take it too); it was tuned on Hexagon, and the review of the +// tuning measured one more clang target: the bench's Cortex-M33 harness +// compiled by clang 18 (thumbv8m.main, -O3) reads 7.1 % / 5.9 % fewer +// instructions at N = 512 / 2048 forced than not. GCC is left to its own +// choices: forced there, the Cortex-M keys with an FPU (M4F, M33, M55 +// without CMSIS) cost 4.5 - 9.1 % more instructions (spills in the larger +// bodies; the soft-float M4 0.1 - 0.2 % less). Not forced when optimizing +// for size (-Os / -Oz define __OPTIMIZE_SIZE__; clang-cl defines it under +// /O1 and /Os): forced, the Hexagon float size probe's .text grows from +// 238,628 to 263,204 bytes. The choice is made per translation unit, so a +// program that mixes size- and speed-optimized translation units gets +// whichever copy of the out-of-line members the linker keeps: "not forced +// under -Os" is a property of a whole-program -Os build, not a guarantee +// per call site. #if defined(__clang__) && !defined(__OPTIMIZE_SIZE__) #define TAP_DSP_SRDIF_INLINE __attribute__((always_inline)) #else @@ -717,16 +727,19 @@ namespace tap::dsp::detail { /// a b - c d, the first product rounded in a statement of its own. /// - /// Without contraction this is the same two products and one - /// subtraction, rounded the same way, as the plain expression (the - /// output bits and the fingerprints do not move). The float spelling + /// Without contraction, and with FLT_EVAL_METHOD == 0, this is the + /// same two products and one subtraction, rounded the same way, as + /// the plain expression (the output bits and the fingerprints do not + /// move; the fingerprint test asserts FLT_EVAL_METHOD == 0). Under + /// excess precision (x87, FLT_EVAL_METHOD == 2) the split statement + /// rounds ab to float where the plain expression would not. The float spelling /// is for the compilers that contract within a statement (clang, /// AppleClang): on `a * b - c * d` clang fuses the left product and /// negates the right one, fma(a, b, -(c d)), which costs Hexagon a /// separate negation in the two slots the float arithmetic itself /// needs; here the statement is `ab - c * d`, which it fuses as /// fma(-c, d, ab), one Hexagon multiply-subtract (Rx -= sfmpy(Rs, - /// Rt)). Hexagon float scenarios -0.65 % / -0.43 % (N = 512 / 2048). + /// Rt)). Hexagon float scenarios -0.65 % / -0.43 % packets (N = 512 / 2048). /// GCC contracts across statements after SSA and compiles both /// spellings the same (the Cortex-M counts are identical). Double /// keeps the plain expression: Hexagon has no double fused @@ -934,7 +947,8 @@ namespace tap::dsp::detail { /// (the output bits do not move). Measured in this tree with register /// leaves of at most 16 / 8 / 4 values restored, against the pairs /// (docs/fft-design.md, "Hexagon (clang) tuning"), millions of - /// instructions: Hexagon rfft_f32_512 46.10 / 45.88 / 45.89 / 45.94, + /// packets on Hexagon and of instructions on the Cortex-M keys: + /// Hexagon rfft_f32_512 46.10 / 45.88 / 45.89 / 45.94, /// rfft_f32_2048 51.79 / 51.58 / 51.55 / 51.61, rfft_f64_512 101.30 / /// 100.63 / 100.12 / 99.94; M4F rfft_f32_512 92.54 / 92.00 / 91.85 / /// 91.55, rfft_f32_2048 106.30 / 105.83 / 105.58 / 105.15; the M33 and