diff --git a/.github/workflows/bench.yml b/.github/workflows/bench.yml index 8450f7f..ca8ce60 100644 --- a/.github/workflows/bench.yml +++ b/.github/workflows/bench.yml @@ -11,6 +11,8 @@ name: Bench # m55 Cortex-M55, Helium (mps3-an547) CMSIS-DSP (deployed profile) # m55-ooura Cortex-M55, TAP_DSP_FFT_CMSIS=OFF (mps3-an547) srdif float32 fallback # (the key's name is historical; baselines are keyed on it) +# hexagon Hexagon v68 + HVX-128, clang 19 (qemu-hexagon user mode) srdif float32 +# and double (a hosted target, so rfft_f64_512 is built and gated too) # # Every key measures what ships through tap::dsp::basic_real_fft: the srdif # engine (fft/srdif.h), or CMSIS-DSP behind the same class on the `m55` key, @@ -53,7 +55,7 @@ on: permissions: contents: read -# The expensive workflow: five QEMU jobs. One run per ref at a time. +# The expensive workflow: six QEMU jobs. One run per ref at a time. concurrency: group: bench-${{ github.workflow }}-${{ github.ref }} cancel-in-progress: true @@ -115,6 +117,20 @@ jobs: text_ceiling_f32: 29056 text_ceiling_q15: 27968 text_ceiling_q31: 27392 + # Hexagon: MuTap's third target (its Hexagon CI leg and ratchet use + # this toolchain, these flags and this QEMU), added when the srdif + # engine was found to cost more instructions there than the engine + # it replaced, which no key here measured (bench/README.md). + # clang, not GCC; hosted (static musl) under qemu-hexagon, which + # the job builds with plugin support. The size probes are measured + # with the toolchain's llvm-size. + - key: hexagon + arch: hexagon + toolchain: cmake/hexagon-linux-musl.cmake + flags: "" + text_ceiling_f32: 245824 + text_ceiling_q15: 251968 + text_ceiling_q31: 251456 env: # Commit the tag pointed at when pinned (tags are movable; commit SHAs # are not); the header's digest is verified on download. This SHA is @@ -125,6 +141,14 @@ jobs: # that is never shipped (tools/qemu_insn_plugin/insn_count.c). QEMU_PLUGIN_HEADER_URL: https://raw.githubusercontent.com/qemu/qemu/11aa0b1ff115b86160c4d37e7c37e6a6b13b77ea/include/qemu/qemu-plugin.h QEMU_PLUGIN_HEADER_SHA256: "c53a2af163e80e3f4bc6c60dbdfc84003db329d757e37cd8a16a77e1d82606ff" + # The hexagon key only: the QEMU release its qemu-hexagon is built + # from (digest-verified, the plugin header's release) and the + # CodeLinaro toolchain (MuTap's pins). + QEMU_SRC_URL: https://download.qemu.org/qemu-8.2.2.tar.xz + QEMU_SRC_SHA256: 847346c1b82c1a54b2c38f6edbd85549edeb17430b7d4d3da12620e2962bc4f3 + HEXAGON_TOOLCHAIN: clang+llvm-19.1.5-cross-hexagon-unknown-linux-musl + HEXAGON_TOOLCHAIN_SHA256: 55b41922318f6331590ab7baa7f5dbdd99c109327a9c44a52c5e9878fab148c1 + HEXAGON_TOOLCHAIN_VERSION: clang version 19.1.5 PLUGIN: /tmp/libinsncount.so BUILD_DIR: build-${{ matrix.key }} SIZE_DIR: build-${{ matrix.key }}-minsizerel @@ -133,18 +157,87 @@ jobs: - uses: actions/checkout@v4 - name: Install Arm toolchain and QEMU + if: matrix.arch != 'hexagon' run: > sudo apt-get update -q && sudo apt-get install -y -q gcc-arm-none-eabi qemu-system-arm libglib2.0-dev pkg-config - # These versions go into bench/README.md next to the numbers when a key - # is seeded or re-recorded; the counts are only comparable within a - # toolchain/QEMU pair. + # Neither Debian's nor CodeLinaro's qemu-hexagon enables TCG plugins, + # so the hexagon key builds its own from the pinned release (the same + # pin and digest as MuTap's hexagon job, cached on the digest). + - name: Install Hexagon build dependencies + if: matrix.arch == 'hexagon' + run: > + sudo apt-get update -q && + sudo apt-get install -y -q libglib2.0-dev pkg-config ninja-build flex bison zstd + + - name: Cache plugin-enabled qemu-hexagon + if: matrix.arch == 'hexagon' + id: qemu-hex + uses: actions/cache@v4 + with: + path: ~/qemu-hexagon-plugins + key: qemu-hexagon-plugins-${{ env.QEMU_SRC_SHA256 }}-1 + + - name: Build plugin-enabled qemu-hexagon + if: matrix.arch == 'hexagon' && steps.qemu-hex.outputs.cache-hit != 'true' + run: | + curl -sfLo /tmp/qemu-src.tar.xz "$QEMU_SRC_URL" + actual=$(sha256sum /tmp/qemu-src.tar.xz | cut -d' ' -f1) + if [ "$actual" != "$QEMU_SRC_SHA256" ]; then + echo "::error::qemu source checksum mismatch"; exit 1 + fi + tar -xJf /tmp/qemu-src.tar.xz -C /tmp + cd /tmp/qemu-*/ + ./configure --target-list=hexagon-linux-user --enable-plugins \ + --disable-docs --disable-tools --disable-system + ninja -C build qemu-hexagon + mkdir -p ~/qemu-hexagon-plugins + cp build/qemu-hexagon ~/qemu-hexagon-plugins/ + + # The compiler determines every hexagon count, so it is pinned like the + # QEMU source: digest-verified on download, and its version asserted. + - name: Download the Hexagon toolchain + if: matrix.arch == 'hexagon' + run: | + set -eo pipefail + curl -fsSLo /tmp/hexagon-toolchain.tar.zst \ + "https://artifacts.codelinaro.org/artifactory/codelinaro-toolchain-for-hexagon/19.1.5/${HEXAGON_TOOLCHAIN}.tar.zst" + echo "$HEXAGON_TOOLCHAIN_SHA256 /tmp/hexagon-toolchain.tar.zst" | sha256sum -c - + mkdir -p "$RUNNER_TEMP/hexagon-toolchain" + tar --zstd -x -C "$RUNNER_TEMP/hexagon-toolchain" -f /tmp/hexagon-toolchain.tar.zst + rm /tmp/hexagon-toolchain.tar.zst + cxx=$(find "$RUNNER_TEMP/hexagon-toolchain" -maxdepth 5 \( -type f -o -type l \) \ + -name 'hexagon-unknown-linux-musl-clang++' | sort | head -1) + test -n "$cxx" || { echo "::error::no hexagon clang++ in the toolchain archive"; exit 1; } + "$cxx" --version | head -1 | grep -qF "$HEXAGON_TOOLCHAIN_VERSION" \ + || { echo "::error::the toolchain is not $HEXAGON_TOOLCHAIN_VERSION"; exit 1; } + echo "HEXAGON_TOOLCHAIN_ROOT=$(dirname "$(dirname "$cxx")")" >> "$GITHUB_ENV" + echo "$HOME/qemu-hexagon-plugins" >> "$GITHUB_PATH" + + # linux-user qemu prints "unknown option 'plugin'" when built without + # plugin support; a missing binary is caught first. + - name: Check qemu-hexagon has plugin support + if: matrix.arch == 'hexagon' + run: | + set -o pipefail + q="$HOME/qemu-hexagon-plugins/qemu-hexagon" + test -x "$q" || { echo "::error::no qemu-hexagon at $q"; exit 1; } + if "$q" -plugin help 2>&1 | grep -q "unknown option"; then + echo "::error::built qemu-hexagon lacks plugin support"; exit 1 + fi + - name: Toolchain versions run: | - arm-none-eabi-gcc --version | head -1 - qemu-system-arm --version | head -1 + set -o pipefail + if [ "${{ matrix.arch }}" = hexagon ]; then + "$HEXAGON_TOOLCHAIN_ROOT/bin/hexagon-unknown-linux-musl-clang++" --version | head -1 + qemu-hexagon --version | head -1 + else + arm-none-eabi-gcc --version | head -1 + qemu-system-arm --version | head -1 + fi cmake --version | head -1 - name: Build counting plugin @@ -187,6 +280,8 @@ jobs: CEILING_Q31: ${{ matrix.text_ceiling_q31 }} run: | set -eo pipefail + SIZE_TOOL=arm-none-eabi-size + if [ "${{ matrix.arch }}" = hexagon ]; then SIZE_TOOL="$HEXAGON_TOOLCHAIN_ROOT/bin/llvm-size"; fi cmake -S . -B "$SIZE_DIR" \ -DCMAKE_BUILD_TYPE=MinSizeRel \ -DCMAKE_TOOLCHAIN_FILE=${{ matrix.toolchain }} \ @@ -198,7 +293,7 @@ jobs: profile=${entry%%:*} ceiling=${entry#*:} probe="$SIZE_DIR/bench/tap_dsp_size_probe_rfft_${profile}_512" - arm-none-eabi-size -A "$probe" | tee "size-$profile.txt" + $SIZE_TOOL -A "$probe" | tee "size-$profile.txt" text=$(awk '$1 == ".text" { print $2 }' "size-$profile.txt") case "$text" in ''|*[!0-9]*) echo "::error::$KEY: $profile probe .text unavailable (no numeric .text row in size-$profile.txt: '$text')"; exit 1 ;; @@ -289,7 +384,7 @@ jobs: # Folds the per-key seeding artifacts (each has one key filled in) into the # single file the seeding commit copies to bench/baselines.json. Fails when # any icount job failed, so a partial seed from a half-red run can never be - # merged. NEVER a required check: the five `icount ` jobs are the + # merged. NEVER a required check: the six `icount ` jobs are the # required checks once seeded (bench/README.md). seed-summary: name: Merge seeding artifacts (never a required check) diff --git a/CLAUDE.md b/CLAUDE.md index dbb4ba8..8d508d2 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -40,8 +40,9 @@ asset's contract summary. each floor a measured number against the double profile. The floating profiles run the srdif engine (`fft/srdif.h`): the N real samples as N/2 complex values, a split-radix decimation-in-frequency kernel of length N/2 (fused two-level passes, compile-time blocks of 32 - and 64, register leaves of 16 and fewer), the bit-reversal permutation and the real post-pass - derived for the fixed-point engine (#39), written clean-room from the literature the header cites + and 64, one level at a time in memory below that, forced inlining under every clang, tuned on + Hexagon), the bit-reversal permutation and the real post-pass derived for the fixed-point engine + (#39), written clean-room from the literature the header cites (tap/DspTap#42; it replaced, at the same contract, the C++20 port of Ooura's `rdft`, `fft/split_radix.h`, that the floating profiles ran from Stage 2b, so DspTap ships no code derived from Ooura's package — the maintainer's judgement, `NOTICE.md`). Since Stage 4 the engine is a @@ -102,7 +103,9 @@ parity suite runs for real. Each leg builds `tap_dsp_tests` one-shot (`tests/bar with the `MAIN_FILTER` in `tests/CMakeLists.txt` naming what the emulation budget excludes, plus the srdif fingerprints, with every FFT sweep capped at `TAP_DSP_TEST_MAX_FFT_N=4096`. The instruction-count ratchet (`bench.yml`, `bench/README.md`) runs the same four cores under five -baseline keys at ±3 %, with `.text` ceilings per profile; a red ratchet is a failing check. The +baseline keys at ±3 %, plus a `hexagon` key (clang 19, v68 + HVX, under `qemu-hexagon`; it +counts packets, not instructions), with `.text` ceilings per profile; a red ratchet is a failing +check. The hosted build and battery need no Arm toolchain; where `gcc-arm-none-eabi` and `qemu-system-arm` are installed, `cmake -S . -B build-m33 -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m33-mps2.cmake` reproduces a leg. diff --git a/README.md b/README.md index 4187ed3..001dde4 100644 --- a/README.md +++ b/README.md @@ -25,7 +25,7 @@ its size range and shareability as contract numbers: | Engine | Profiles | Selected by | Size range | Shareable | What it is | |---|---|---|---|---|---| | srdif (`fft/srdif.h`) | `double`; `float` unless a backend is on | default | 4 … 2^30 | yes | a split-radix decimation-in-frequency kernel of length N/2 (Duhamel–Hollmann 1984; Sorensen–Heideman–Burrus 1986), the bit-reversal permutation and the real post-pass of Cooley–Lewis–Welch 1970 / Sorensen et al. 1987 (the fixed-point engine's structure in floating point), written from the literature; exactly the split-radix operation count, 2N log₂N − 2N − 2 real operations per forward transform; tables from integer arithmetic (no libm), built in the constructor, so without fp-contraction the output bits are one row for every compiler and target CI runs (pinned as output fingerprints in `tests/test_fft_srdif_fingerprint.cpp`). Since tap/DspTap#42, written clean-room; it replaced, at the same contract, the C++20 port of Ooura's `rdft` (`fft/split_radix.h`) that the floating profiles ran from Stage 2b (#31) until #42 (see Provenance) | -| CMSIS-DSP Helium (`fft/backends/cmsis.h`) | `float` | `TAP_DSP_FFT_CMSIS` (default ON exactly when the compiler targets floating-point Helium, `__ARM_FEATURE_MVE & 2`: the Cortex-M55 / M85 class; OFF for every other target, other Cortex-M cores included) | **32 … 4096** (CMSIS-DSP's own init table; the constructor checks the init status since Stage 4, in a debug build — outside this range the library never initializes, and a release build is undefined behaviour with no fault promised: measured under QEMU, N = 16 and 8192 give a wrong spectrum silently, N = 4 a wrong spectrum and a corrupted heap) | no | Arm's radix-4/8 MVE real FFT, re-presented in the same contract to float epsilon; on the M55 the srdif engine (the `m55-ooura` icount key) executes 1.60× (N = 512) / 1.78× (N = 2048) the instructions of the CMSIS build (the `m55` key) over the whole ratchet scenario, transform plus the class's copy loop, 2/N scaling and checksum (`bench/README.md`) | +| CMSIS-DSP Helium (`fft/backends/cmsis.h`) | `float` | `TAP_DSP_FFT_CMSIS` (default ON exactly when the compiler targets floating-point Helium, `__ARM_FEATURE_MVE & 2`: the Cortex-M55 / M85 class; OFF for every other target, other Cortex-M cores included) | **32 … 4096** (CMSIS-DSP's own init table; the constructor checks the init status since Stage 4, in a debug build — outside this range the library never initializes, and a release build is undefined behaviour with no fault promised: measured under QEMU, N = 16 and 8192 give a wrong spectrum silently, N = 4 a wrong spectrum and a corrupted heap) | no | Arm's radix-4/8 MVE real FFT, re-presented in the same contract to float epsilon; on the M55 the srdif engine (the `m55-ooura` icount key) executes 1.60× (N = 512) / 1.77× (N = 2048) the instructions of the CMSIS build (the `m55` key) over the whole ratchet scenario, transform plus the class's copy loop, 2/N scaling and checksum (`bench/README.md`) | | Apple vDSP (`fft/backends/accelerate.h`) | `float` | `TAP_DSP_FFT_ACCELERATE` (default ON on Apple) | 4 … 2^20 | no | `vDSP_fft_zrip`, same contract to float epsilon; ~3× faster per transform on Apple Silicon is MuTap's transform-only measurement against the engine the library carried before (tap/MuTap#31), not re-measured in this repo against srdif (the same-binary parity *test* exists since Stage 4; the microbenchmark needs a Mac) | | int32 radix-4 (`fft/fixed_point.h`) | Q15, Q31 | the sample type (the second template argument is the scaling policy here) | 4 … 65536 | Q31 yes, Q15 no | one kernel over Q1.30 twiddles under two scaling policies, returning an exponent | diff --git a/bench/README.md b/bench/README.md index 75e4af6..c5d38f4 100644 --- a/bench/README.md +++ b/bench/README.md @@ -78,10 +78,12 @@ integer FNV-1a-64 fold over its bit pattern (`bench_common.h`), printed at the end: ``` -TAP_DSP_ICOUNT_DONE ok=1 engine=basic_real_fft backend=srdif scenario=rfft_f32_512 checksum=0xd26ebf9b45534325 +TAP_DSP_ICOUNT_DONE ok=1 engine=basic_real_fft backend=srdif scenario=rfft_f32_512 checksum=0x322a64b478d14325 ``` -(the `m4f` key's line on the srdif engine). The fold is exact and +(the `m4f` key's line on the srdif engine since its Hexagon tuning; +`0xd26ebf9b45534325` from #42 until then: the same outputs at +`-ffp-contract=off`, fused differently by GCC). The fold is exact and order-sensitive: two runs of the same binary print the same value, and a 1-ulp change in any single output changes it (verified on the port, the engine of the time, by nudging one spectrum bin at one iteration with @@ -145,6 +147,26 @@ audit Part 3); the outcome is in the table below. | `m33` | Cortex-M33, no MVE | `mps2-an505` | srdif engine | | `m55` | Cortex-M55, Helium | `mps3-an547` | CMSIS-DSP — the deployed profile | | `m55-ooura` | Cortex-M55, `-DTAP_DSP_FFT_CMSIS=OFF` | `mps3-an547` | srdif engine — the fallback (the key keeps the historical name of the vendored C the ratchet was seeded on) | +| `hexagon` | Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl | `qemu-hexagon` user mode (8.2.2, built with plugins) | srdif engine; hosted, so `rfft_f64_512` is built and gated too | + +The `hexagon` key counts **packets**, not instructions: qemu-hexagon reports +one guest instruction per VLIW packet (up to four instructions), as the +per-address profile of the Hexagon tuning pass showed (`docs/fft-design.md`, +"Hexagon (clang) tuning"). So on that key the ratchet rewards what the +compiler packs side by side as well as what it executes. The key was added +when MuTap's Hexagon ratchet found the srdif engine 12–13 % above the port on +its chain workloads, which no key here had measured. Its counts are +comparable with MuTap's Hexagon ratchet: same toolchain, flags and QEMU. +Absolute counts carry a per-process offset: user-mode qemu hands the guest +its environment and the binary's path, and the count moves with them. A +44-character longer path read +731 packets, and `env -i` read −1,584. The +offset is constant across the scenarios of one setup. Local setups have read ++473 to +1,723 against the runner, and CI runs agree with each other to the +packet. Compare within a run, and take baselines from CI only (as for every +key). The predecessor cells in the `hexagon` rows below are +local measurements converted to the runner's offset, ±28 packets: a rebuild +of `7a58ebe` at another path read its float scenarios 28 below them and its +fixed-point scenarios equal to #42's. JSON carries no comments, so the provenance of every recorded set lives here, in the table below: the `main` run that recorded it, and the GCC and QEMU @@ -203,6 +225,20 @@ shape is fixed so the record stays greppable: | m33 | rfft_f32_512 | 100,945,841 | 92,979,212 | −7.89 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; confirmed to the instruction by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) (compare mode, +0.00 %) | `72977aa` (the #42 squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_2048 | 102,784,096 | 97,602,563 | −5.04 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; confirmed to the instruction by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) (compare mode, +0.00 %) | `72977aa` (the #42 squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | | m55-ooura | rfft_f32_512 | 89,276,321 | 83,933,381 | −5.98 % | **re-recorded at the srdif replacement** (tap/DspTap#42): the floating engine behind `basic_real_fft` on these keys is now `fft/srdif.h` (a split-radix DIF kernel of length N/2 written from the literature), which replaced the port of Ooura's `rdft` (`fft/split_radix.h`, deleted at #42); every float output bit changes (new checksums) and the counts fall on every key, beyond the −3 % band on every key but m4-softfp, so the exact new counts are recorded (Part 11 / D11). The counts are those of the fix pass for review A of #42 (the bit-reversal permutation without an index table, GCC's SLP vectorizer off for the engine; output bits unchanged), 0.04 … 2.62 % below the first srdif record (1,815,063,718 / 2,244,317,934 m4-softfp, 94,003,130 / 108,030,219 m4f, 94,382,323 / 108,470,414 m33, 86,188,216 / 100,091,803 m55-ooura at 512 / 2048). Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: measured inside the band (fixed point +0.49 … +1.15 %, CMSIS +0.00 %) with no change to that code. `docs/fft-design.md`, "The floating engine (srdif)" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-27; confirmed to the instruction by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) (compare mode, +0.00 %) | `72977aa` (the #42 squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4-softfp | rfft_f32_2048 | 2,243,212,021 | 2,241,935,605 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4-softfp | rfft_f32_512 | 1,814,303,702 | 1,813,126,102 | −0.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4f | rfft_f32_2048 | 106,282,752 | 105,151,232 | −1.06 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m4f | rfft_f32_512 | 92,575,609 | 91,545,465 | −1.11 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m33 | rfft_f32_2048 | 106,773,786 | 106,248,986 | −0.49 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m33 | rfft_f32_512 | 92,979,212 | 92,624,908 | −0.38 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m55-ooura | rfft_f32_2048 | 97,602,563 | 97,109,507 | −0.51 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| m55-ooura | rfft_f32_512 | 83,933,381 | 83,628,229 | −0.36 % | **re-recorded at the Hexagon tuning of the srdif engine**: `fft/srdif.h` was tuned for the `hexagon` key (clang 19; forced inlining under clang, the post-pass's two bin pairs loaded before either is stored, the small blocks' split-radix levels in memory down to pairs instead of register leaves, one run-time group per loop step in double, a multiply-subtract spelling for clang's contraction), and the float count falls on every key that runs the engine, −0.06 … −1.11 %, inside the band; the exact new counts are recorded (bench/README.md policy: the winning commit re-records). No output bit moves at `-ffp-contract=off` (the srdif fingerprints pass unchanged on the host, the four QEMU legs and Hexagon); at the bench's default flags the FPU keys' checksums move (GCC contracts across the former leaves differently), `m4-softfp`'s do not. Fixed-point scenarios and the `m55` (CMSIS) key are not re-recorded: unchanged to the instruction. `docs/fft-design.md`, "Hexagon (clang) tuning" | measured locally with `scripts/icount.py` exactly as `bench.yml` runs it (fresh Release build per key, TAP_DSP_BUILD_TESTS=OFF, the pinned plugin header), 2026-09-28; confirmed to the instruction by the `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, compare mode, +0.00 %) | **pending** (the squash) | 13.2.1 (15:13.2.rel1-2) | 8.2.2 (1:8.2.2+ds-0ubuntu1.18) | +| hexagon | rfft_f32_2048 | — | 51,608,903 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 65,684,295 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 54,834,714 (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_f32_512 | — | 45,934,423 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 57,659,223 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 49,732,931 (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_f64_512 | — | 99,939,789 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 110,683,597 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 103,239,537 (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q15_512 | — | 159,976,886 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 159,976,886 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 159,976,886, the same code (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q31_2048 | — | 184,890,581 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 184,890,581 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 184,890,581, the same code (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | +| hexagon | rfft_q31_512 | — | 157,303,237 | — | **seeded** (the `hexagon` key is new): Hexagon v68 + HVX-128, CodeLinaro clang 19.1.5, static musl, under qemu-hexagon user mode; the key counts packets, not instructions (above). Seeded on the tuned srdif engine; the engine as #42 shipped it read 157,303,237 on the first seeding run ([run 36357132506](https://github.com/tap/DspTap/actions/runs/36357132506), `28befbd`), and the engine it replaced 157,303,237, the same code (local, the same harness, converted to the runner's offset; float ±28). `docs/fft-design.md`, "Hexagon (clang) tuning" | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) on the branch (`2a5fc69`, seed mode) | **pending** (the squash) | clang 19.1.5 (CodeLinaro, hexagon-unknown-linux-musl) | qemu-hexagon 8.2.2 (built from the release with `--enable-plugins`) | `before` is `—` for a seed. `main SHA` is the commit on `main` whose push run measured the numbers: a pull-request head SHA stops resolving after @@ -233,7 +269,7 @@ is the seeded baseline): One row per (key, probe) each time a ceiling is recorded or re-recorded; the policy is under "Size runs in the same job" below. `.text` is the row of -`arm-none-eabi-size -A` on the MinSizeRel probe; the ceiling is what +`arm-none-eabi-size -A` (the toolchain's `llvm-size -A` on the `hexagon` key) on the MinSizeRel probe; the ceiling is what `bench.yml` carries for that key and profile. | key | probe | `.text` (bytes) | ceiling | reason | run (URL) | arm-none-eabi-gcc | @@ -257,6 +293,9 @@ policy is under "Size runs in the same job" below. `.text` is the row of | m4f | `rfft_f32_512` | 28,921 | 29,824 | **re-recorded at the srdif replacement** (tap/DspTap#42) (ceiling 45,952 before): measured + 3 %, rounded up to 64 bytes; the srdif engine builds its tables from integer arithmetic, so the float probe no longer links libm's double `sin` / `cos`; the figure is the fix pass's for review A of #42 (no swap-list builder, one table allocation), 1,112 … 1,136 B below the first srdif record | measured locally as `bench.yml` builds the probe (MinSizeRel, `size -A` `.text`), 2026-09-27; confirmed to the byte by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) on `72977aa` | 13.2.1 (15:13.2.rel1-2) | | m33 | `rfft_f32_512` | 28,305 | 29,184 | **re-recorded at the srdif replacement** (tap/DspTap#42) (ceiling 45,376 before): measured + 3 %, rounded up to 64 bytes; the srdif engine builds its tables from integer arithmetic, so the float probe no longer links libm's double `sin` / `cos`; the figure is the fix pass's for review A of #42 (no swap-list builder, one table allocation), 1,112 … 1,136 B below the first srdif record | measured locally as `bench.yml` builds the probe (MinSizeRel, `size -A` `.text`), 2026-09-27; confirmed to the byte by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) on `72977aa` | 13.2.1 (15:13.2.rel1-2) | | m55-ooura | `rfft_f32_512` | 28,161 | 29,056 | **re-recorded at the srdif replacement** (tap/DspTap#42) (ceiling 40,512 before): measured + 3 %, rounded up to 64 bytes; the srdif engine builds its tables from integer arithmetic, so the float probe no longer links libm's double `sin` / `cos`; the figure is the fix pass's for review A of #42 (no swap-list builder, one table allocation), 1,112 … 1,136 B below the first srdif record | measured locally as `bench.yml` builds the probe (MinSizeRel, `size -A` `.text`), 2026-09-27; confirmed to the byte by the push-to-`main` [run 36348787347](https://github.com/tap/DspTap/actions/runs/36348787347) on `72977aa` | 13.2.1 (15:13.2.rel1-2) | +| hexagon | `rfft_f32_512` | 238,628 | 245,824 | **recorded for the new key**: measured + 3 %, rounded up to 64 bytes; `llvm-size -A` of the statically linked probe, so the figure includes musl libc and libc++ (most of it), not DspTap alone | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) (`2a5fc69`; MinSizeRel) | clang 19.1.5 (CodeLinaro) | +| hexagon | `rfft_q15_512` | 244,580 | 251,968 | **recorded for the new key**: measured + 3 %, rounded up to 64 bytes; `llvm-size -A` of the statically linked probe, so the figure includes musl libc and libc++ (most of it), not DspTap alone | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) (`2a5fc69`; MinSizeRel) | clang 19.1.5 (CodeLinaro) | +| hexagon | `rfft_q31_512` | 244,132 | 251,456 | **recorded for the new key**: measured + 3 %, rounded up to 64 bytes; `llvm-size -A` of the statically linked probe, so the figure includes musl libc and libc++ (most of it), not DspTap alone | `workflow_dispatch` [run 36363710480](https://github.com/tap/DspTap/actions/runs/36363710480) (`2a5fc69`; MinSizeRel) | clang 19.1.5 (CodeLinaro) | ### Seeding, and how the job decides what to do @@ -322,7 +361,7 @@ python3 scripts/icount.py --merge a.json b.json # fold per-key files into one regression hide in the slack — the winning commit re-records. - **The ratchet runs on every pull request and on every push to `main`** on every QEMU leg (one run per ref at a time). A red ratchet is a failing - check, not a warning. Once seeded, the five `icount ` jobs are the + check, not a warning. Once seeded, the six `icount ` jobs are the required checks; the artifact-merge job never is. - **`--update` is a written commit on its own**: the measured before/after per key and the reason go into the table above, in its fixed shape. An expected diff --git a/bench/baselines.json b/bench/baselines.json index 662d7c5..bb20fce 100644 --- a/bench/baselines.json +++ b/bench/baselines.json @@ -1,21 +1,29 @@ { + "hexagon": { + "rfft_f32_2048": 51608903, + "rfft_f32_512": 45934423, + "rfft_f64_512": 99939789, + "rfft_q15_512": 159976886, + "rfft_q31_2048": 184890581, + "rfft_q31_512": 157303237 + }, "m33": { - "rfft_f32_2048": 106773786, - "rfft_f32_512": 92979212, + "rfft_f32_2048": 106248986, + "rfft_f32_512": 92624908, "rfft_q15_512": 752189619, "rfft_q31_2048": 875801377, "rfft_q31_512": 714626257 }, "m4-softfp": { - "rfft_f32_2048": 2243212021, - "rfft_f32_512": 1814303702, + "rfft_f32_2048": 2241935605, + "rfft_f32_512": 1813126102, "rfft_q15_512": 746311273, "rfft_q31_2048": 869187566, "rfft_q31_512": 709161574 }, "m4f": { - "rfft_f32_2048": 106282752, - "rfft_f32_512": 92575609, + "rfft_f32_2048": 105151232, + "rfft_f32_512": 91545465, "rfft_q15_512": 753215637, "rfft_q31_2048": 876486909, "rfft_q31_512": 715019447 @@ -28,8 +36,8 @@ "rfft_q31_512": 657595969 }, "m55-ooura": { - "rfft_f32_2048": 97602563, - "rfft_f32_512": 83933381, + "rfft_f32_2048": 97109507, + "rfft_f32_512": 83628229, "rfft_q15_512": 684157511, "rfft_q31_2048": 806142635, "rfft_q31_512": 657595969 diff --git a/cmake/hexagon-linux-musl.cmake b/cmake/hexagon-linux-musl.cmake new file mode 100644 index 0000000..20a92b6 --- /dev/null +++ b/cmake/hexagon-linux-musl.cmake @@ -0,0 +1,63 @@ +# SPDX-License-Identifier: MIT +# Copyright 2026 Timothy Place and the DspTap contributors. +# Adapted from MuTap's cmake/hexagon-linux-musl.cmake (MIT, MuTap contributors). +# +# Cross-compile DspTap for Qualcomm Hexagon: triple hexagon-unknown-linux-musl, +# built with the CodeLinaro "toolchain for hexagon" (clang + musl sysroot + LLVM +# runtimes). Point HEXAGON_TOOLCHAIN_ROOT (cache variable or environment) at +# the unpacked clang+llvm-*-cross-hexagon-unknown-linux-musl/x86_64-linux-gnu +# directory. Used by bench.yml's `hexagon` key (the instruction-count ratchet +# under qemu-hexagon user-mode emulation); MuTap's Hexagon CI leg and ratchet +# use the same toolchain, flags and QEMU, so the two repositories' Hexagon +# counts are comparable. +# +# A hosted Linux target, not bare metal: the bench binaries print through +# stdio and exit normally, and are linked statically (first-class with musl) +# so the emulator needs no sysroot path. +# +# Flags (MuTap's, kept identical): +# -mv68 the newest revision the shipped sysroot libraries +# and qemu-hexagon agree on. +# -mhvx -mhvx-length=128b the 128-byte HVX unit, for auto-vectorization. + +set(CMAKE_SYSTEM_NAME Linux) +set(CMAKE_SYSTEM_PROCESSOR Hexagon) + +if(NOT DEFINED HEXAGON_TOOLCHAIN_ROOT OR HEXAGON_TOOLCHAIN_ROOT STREQUAL "") + set(HEXAGON_TOOLCHAIN_ROOT "$ENV{HEXAGON_TOOLCHAIN_ROOT}") +endif() +if(HEXAGON_TOOLCHAIN_ROOT STREQUAL "") + message(FATAL_ERROR + "Set HEXAGON_TOOLCHAIN_ROOT (cache variable or environment) to the unpacked CodeLinaro " + "hexagon toolchain (clang+llvm-*-cross-hexagon-unknown-linux-musl/x86_64-linux-gnu)") +endif() +# try_compile projects re-read this file without the cache: pass the root on. +set(ENV{HEXAGON_TOOLCHAIN_ROOT} "${HEXAGON_TOOLCHAIN_ROOT}") + +set(TAP_DSP_HEXAGON_TRIPLE hexagon-unknown-linux-musl) +if(EXISTS "${HEXAGON_TOOLCHAIN_ROOT}/bin/${TAP_DSP_HEXAGON_TRIPLE}-clang++") + set(CMAKE_C_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/${TAP_DSP_HEXAGON_TRIPLE}-clang") + set(CMAKE_CXX_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/${TAP_DSP_HEXAGON_TRIPLE}-clang++") +else() + set(CMAKE_C_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/clang") + set(CMAKE_CXX_COMPILER "${HEXAGON_TOOLCHAIN_ROOT}/bin/clang++") + set(CMAKE_C_COMPILER_TARGET ${TAP_DSP_HEXAGON_TRIPLE}) + set(CMAKE_CXX_COMPILER_TARGET ${TAP_DSP_HEXAGON_TRIPLE}) +endif() + +set(TAP_DSP_HEXAGON_ARCH_FLAGS "-mv68 -mhvx -mhvx-length=128b") +set(CMAKE_C_FLAGS_INIT "${TAP_DSP_HEXAGON_ARCH_FLAGS}") +set(CMAKE_CXX_FLAGS_INIT "${TAP_DSP_HEXAGON_ARCH_FLAGS}") +# --eh-frame-hdr: not implied for -static links, but the unwinder needs +# PT_GNU_EH_FRAME to find the exception tables. +set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -Wl,--eh-frame-hdr") + +set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER) +set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY) + +find_program(TAP_DSP_QEMU_HEXAGON NAMES qemu-hexagon-static qemu-hexagon) +if(TAP_DSP_QEMU_HEXAGON) + set(CMAKE_CROSSCOMPILING_EMULATOR "${TAP_DSP_QEMU_HEXAGON}") +endif() diff --git a/docs/fft-design.md b/docs/fft-design.md index 2853103..732388e 100644 --- a/docs/fft-design.md +++ b/docs/fft-design.md @@ -942,7 +942,7 @@ host. The port and its header were deleted at tap/DspTap#42 (hand-off commit `17db855`), and the srdif engine that replaced it is not a transliteration of anything: its own constraints (statement order under D9, the integer -table generator, the register leaves) are stated in `fft/srdif.h`. The rules +table generator, the leaf blocks) are stated in `fft/srdif.h`. The rules below applied to the port from Stage 2a (tap/DspTap#28) to #42. They lived in the engine header (`include/tap/dsp/fft/split_radix.h`) so the next person would not undo them, and are recorded here because they were what the bit @@ -1926,7 +1926,10 @@ the kernel is arranged around that: unrolled groups) from their own small tables; blocks of 16 and fewer are leaves held in registers through every remaining level. The run-time recursion handles l ≥ 128 only, and its groups run two per loop step so - the eight row pointers serve both. + the eight row pointers serve both. (As #42 shipped it. The Hexagon tuning + below replaced the register leaves with one split-radix level at a time in + memory down to pairs, and runs the double profile's run-time groups one + per step.) - **The permutation** keeps no index table: a bit-reversed counter walks beside the natural index (the reverse-carry increment of Gold and Rader, *Digital Processing of Signals*, 1969: adding one flips an index's @@ -2293,6 +2296,212 @@ made. `FloatEngineTracksDoubleAtN512`, 2.25e-7 against 9.73e-8 / 1.05e-7; `DoubleForwardTracksCompensatedDft`, 7.4e-16 against 1.49–1.70e-16). + +### Hexagon (clang) tuning + +The engine met "must not regress" on every key the ratchet measured at #42, +all four Cortex-M cores under GCC 13.2. DspTap's main consumer also ships on +Qualcomm Hexagon (clang 19, v68 with HVX-128), where the srdif engine of +`72977aa` executed more instructions than the engine it replaced; the bench +gained a `hexagon` key for it (`cmake/hexagon-linux-musl.cmake`, the +CodeLinaro clang 19.1.5 toolchain, `-mv68 -mhvx -mhvx-length=128b`, static +musl, a plugin-enabled `qemu-hexagon` 8.2.2), and the engine was tuned until +every floating Hexagon scenario is below the predecessor's count, without +costing any Cortex-M key. Done under the same clean-room rules as the +engine: from this tree, its own disassembly and counts, and the literature +the header cites. + +**What the Hexagon key counts.** `qemu-hexagon` translates a packet (up to +four instructions issued together) as one guest instruction, so the plugin +counts **packets**, not instructions: a per-address profile (a scratch +plugin, one counter per translated instruction) holds counts at +packet-start addresses only, and they sum to the ratchet's total. The +Hexagon count therefore rewards what the scheduler can pack side by side, +not only what it executes. In the disassembly the float and double +arithmetic takes two of a packet's four slots at most (with the negations +and shifts that share them), the loads and stores the other two; a double +multiply is six instructions (`dfmpyfix` twice, `dfmpyll`, `dfmpylh` twice, +`dfmpyhh`), a double add one. + +**Where the count went at `72977aa`** (`rfft_f32_512`, 57.66 M packets): +the class's out-of-place copies, which clang compiles to calls of musl's +`memcpy`, 17.0 M (29 %; 33.7 M of the double scenario's 110.7 M), and the +harness's fold 6.1 M, both the same for either engine (the fixed-point +scenarios, which do not run srdif, measure the same on both trees); the +engine the rest. clang's inliner had kept the butterflies' callers out of +line: the fused pass called `group<…>` per group, each register leaf went +through a stack array of values (`leaf_blocks(cv*)`) and a call, and the +index-sequence lambdas of `block<32 | 64>` and `leaf` were calls. GCC +inlines differently and the Cortex-M keys had not shown it. + +**Measured** with `scripts/icount.py` as `bench.yml` runs it (a fresh +Release build per key), on one machine, 2026-09-28. The Hexagon counts carry +a per-process offset: user-mode qemu passes the guest its environment and the +binary's path, and the count moves with them (the review of the tuning +measured +731 packets for a 44-character longer path, and −1,584 under +`env -i`). The offset is constant across the scenarios of one setup. Here it +read +1,250 against the setup that measured the predecessor (the fixed-point +scenarios, which do not run srdif, read the same up to that offset), so the +predecessor column below is that setup's count + 1,250. The review rebuilt +`7a58ebe` at its own path and read the predecessor's float scenarios 28 +packets below these cells, so read them as ±28. No percentage moves. +Against CI the offsets were +1,723 (this setup) and +1,596 (the review's). +The Cortex-M float counts reproduce `bench/baselines.json` exactly. The +fixed-point rows read the same +0.49 … +1.15 % in-band drift as on `main`: +they were seeded at #32 and have not been re-recorded since. + +| key | scenario | `72977aa` | tuned | Δ | predecessor | tuned vs predecessor | +|---|---|---:|---:|---:|---:|---:| +| `hexagon` | `rfft_f32_512` | 57,660,946 | 45,936,146 | −20.33 % | 49,734,654 | −7.64 % | +| `hexagon` | `rfft_f32_2048` | 65,686,018 | 51,610,626 | −21.43 % | 54,836,437 | −5.88 % | +| `hexagon` | `rfft_f64_512` | 110,685,320 | 99,941,512 | −9.71 % | 103,241,260 | −3.20 % | +| `m4-softfp` | `rfft_f32_512` | 1,814,303,702 | 1,813,126,102 | −0.06 % | 1,864,929,141 | −2.78 % | +| `m4-softfp` | `rfft_f32_2048` | 2,243,212,021 | 2,241,935,605 | −0.06 % | 2,294,360,259 | −2.28 % | +| `m4f` | `rfft_f32_512` | 92,575,609 | 91,545,465 | −1.11 % | 97,138,544 | −5.76 % | +| `m4f` | `rfft_f32_2048` | 106,282,752 | 105,151,232 | −1.06 % | 111,278,416 | −5.51 % | +| `m33` | `rfft_f32_512` | 92,979,212 | 92,624,908 | −0.38 % | 100,945,841 | −8.24 % | +| `m33` | `rfft_f32_2048` | 106,773,786 | 106,248,986 | −0.49 % | 115,465,626 | −7.98 % | +| `m55-ooura` | `rfft_f32_512` | 83,933,381 | 83,628,229 | −0.36 % | 89,276,321 | −6.33 % | +| `m55-ooura` | `rfft_f32_2048` | 97,602,563 | 97,109,507 | −0.51 % | 102,784,096 | −5.52 % | + +Unchanged, to the instruction: the `m55` key (CMSIS-DSP, 52,382,329 / +54,858,158) and every fixed-point scenario on every key, Hexagon included +(159,978,609 / 157,304,960 / 184,892,304 for Q15 512 / Q31 512 / Q31 2048). +The Cortex-M float baselines are re-recorded to these counts +(`bench/README.md`); the Hexagon key is seeded from CI. Against CMSIS-DSP +on the M55 the srdif engine now executes 1.60× (N = 512) / 1.77× +(N = 2048) the instructions (1.60× / 1.78× before). + +**What paid, step by step** (Hexagon, each step on top of the one before; +`rfft_f32_512` / `rfft_f32_2048` / `rfft_f64_512`, millions of packets): + +| step | f32 512 | f32 2048 | f64 512 | +|---|---:|---:|---:| +| `72977aa` | 57.66 | 65.69 | 110.69 | +| clang only: butterflies, groups, leaves, post-pass pairs and swaps forced inline (`TAP_DSP_SRDIF_INLINE`) | 49.67 | 55.24 | 105.83 | +| post-pass: two bin pairs with every load before any store (`post_two`) | 48.26 | 53.80 | 104.54 | +| the first level of a block of 16 (and of 8, 4 in double) in memory, register leaves of 8 (float) / 2 (double) | 48.15 | 53.70 | 103.02 | +| forced inline also the index-sequence lambdas and the compile-time blocks | 46.10 | 51.72 | 100.44 | +| double: the run-time fused pass one group per loop step | 46.10 | 51.72 | 100.06 | +| `mul_sub`: `a b − c d` spelled so clang fuses it as a multiply-subtract | 45.88 | 51.58 | 100.09 | +| no register leaves at all: levels in memory down to pairs, both profiles | 45.94 | 51.61 | 100.08 | +| `mul_sub`'s split spelling for float only | 45.94 | 51.61 | 99.94 | + +Each change measured alone against the final tree (the change undone, +everything else kept; Hexagon f32 512 / f32 2048 / f64 512, then the +Cortex-M keys): + +- **Forced inlining (clang only).** Without it: 54.13 / 62.12 / 107.12 M + (+17.8 / +20.4 / +7.2 %). Forced under GCC as well: `m4f` +9.1 / +6.8 %, + `m33` +5.7 / +4.5 %, `m55-ooura` +6.3 / +4.9 % (spills in the larger + bodies), `m4-softfp` −0.22 / −0.14 %; so GCC keeps its own choices. Not + forced under `-Os` / `-Oz` (`__OPTIMIZE_SIZE__`): forced, the Hexagon + MinSizeRel float probe's `.text` grows from 238,628 to 263,204 bytes. +- **`post_two`.** Pair by pair the compiler cannot move the second pair's + loads above the first pair's stores (a store through `pk` may alias a + load through `pj` as far as it knows), so the two pairs' arithmetic + cannot share packets. Without it: 47.47 / 53.17 / 100.97 M (+3.3 / +3.0 / + +1.0 %). The Cortex-M keys pay 0.00 – 0.02 % for it. +- **Levels in memory down to pairs.** Register leaves of at most 16 / 8 / 4 + values restored in the final tree: Hexagon 46.10 / 45.88 / 45.89 (f32 512), + 51.79 / 51.58 / 51.55 (f32 2048), 101.30 / 100.63 / 100.12 (f64), against + 45.94 / 51.61 / 99.94 for pairs; `m4f` 92.54 / 92.00 / 91.85 and 106.30 / + 105.83 / 105.58 against 91.55 / 105.15; `m33` and `m55-ooura` 0.3 – 0.5 % + below leaves of 8, `m4-softfp` flat. A leaf of 16 holds 32 values: the + M4F's whole single-precision register file and all 32 of Hexagon's general + registers, of which a double takes two. Pairs cost Hexagon float 0.1 % + against its best (leaves of 4 or 8) and are the best everywhere else, in + one code path. +- **One group per step in double.** Two groups per step: 100.32 M (+0.38 %). + One per step in float too: 46.16 / 52.12 M (+0.48 / +0.99 %), so float + keeps two. +- **`mul_sub`.** clang contracts within a statement and, on `a * b − c * d`, + fuses the left product and negates the right, `fma(a, b, −(c d))`: a + separate negation (`togglebit`) in the two slots the float arithmetic + needs. `ab − c * d` with `ab` a statement of its own fuses as + `fma(−c, d, ab)`, one `Rx −= sfmpy(Rs, Rt)`. Without it: float 46.24 / + 51.83 M (+0.65 / +0.43 %). Double keeps the plain expression (Hexagon has + no double fused multiply-add; the split spelling only moved the schedule, + +0.14 %). Without contraction both spellings are the same two products + and one subtraction; GCC, which contracts after SSA across statements, + compiles both the same (identical Cortex-M counts). + +Tried and not kept: + +- All eight loads of a group before its stores (the level-l butterfly at + j + l/8 no longer waits for the stores of the one at j): +1.8 / +3.2 % + float, +0.1 % double on Hexagon at the time; the extra live values cost + more than the freedom. +- The permutation's six unconditional exchanges as two batches of three, + every load before any store (measured at the `mul_sub` step): Hexagon + float −0.39 / −0.37 %, but double +0.34 % (−0.05 % in batches of two), + `m55-ooura` +1.69 / +1.47 % and `m4-softfp` +0.10 / +0.08 %. +- Register-leaf sizes of 16, 8 and 4 (above). + +**Output bits.** Unchanged wherever the fingerprints are defined: every +change reorders loads, stores, calls and loop steps, or (`mul_sub`) splits a +statement without changing its operations, so without contraction every +output is the same operations in the same order. `tap_dsp_srdif_fingerprint` +(`-ffp-contract=off`) passes unchanged on x86-64 g++ 13.3 and clang++ 18.1, +on the four QEMU legs, and on Hexagon (clang 19.1.5 under `qemu-hexagon`, +every N from 4 to 65536, both profiles; at `72977aa` as well): **Hexagon +matches the one row**, though CI does not run it there. Under default flags, +which contract, the bench checksums move where an FMA exists: the Cortex-M +FPU keys' `rfft_f32_512` / `rfft_f32_2048` lines (`m4f`, `m33` and +`m55-ooura` alike) from `0xd26ebf9b45534325` / `0xd06aa150d5bd9325` to +`0x322a64b478d14325` / `0xb7eeda0471060d25` (GCC fuses across the former +register leaves differently once their levels run through memory; `post_two` +and `mul_sub` alone leave the GCC lines as they were), and Hexagon float +from `0x8407c06ac8ac8325` / `0xe3dffd98aa4cf725` to `0x46adbb5e3ebf2325` / +`0x7ecf6080a0f94125` (`mul_sub`). Without an FMA nothing moves: x86-64 +(`0x5caf505901a97f25`, g++ and clang++), `m4-softfp`, Hexagon double. The +error statistics are the contraction policy's business ("The fp-contraction +policy"); no accuracy re-run was needed, since no bit moved at +`-ffp-contract=off`. + +**Contract.** Unchanged: packing, sign, scale, sizes 4 … 2^30, +shareability, `noexcept` and allocation-free transforms, the tables and the +heap formula (nothing about the tables changed). The Hexagon test battery +(`tap_dsp_tests` and the side targets built with the Hexagon toolchain file, +run by ctest through `qemu-hexagon`, sweeps capped at 4096) passes 423 of +426; the three failures are `fft_abi_tag`'s `dlopen` tests, which a static +musl image cannot run ("Dynamic loading not supported"). + +**`.text`** of the MinSizeRel float probe: `m4-softfp` 31,465 B (31,081 +before), `m4f` 28,985 (28,921), `m33` 28,385 (28,305), `m55-ooura` 28,289 +(28,161), all under their ceilings, which are not re-recorded; Hexagon +238,628 (239,044); the Q15 / Q31 probes and the `m55` CMSIS probe +unchanged. + +**Host timings** (informational; `bench/bench_fft.cpp`, `-O3 -DNDEBUG`, +g++ 13.3 and clang++ 18.1, Xeon 2.8 GHz VM (4 vCPUs, load about 1), pinned to one core, 11 runs +alternating the two trees, the median of the per-run minima, ns per +transform; `72977aa` → tuned): + +| compiler | scenario | forward | inverse | +|---|---|---:|---:| +| g++ | `rfft_f32_512` | 1,714 → 1,682 (−1.9 %) | 1,788 → 1,780 (−0.5 %) | +| g++ | `rfft_f32_2048` | 8,259 → 8,022 (−2.9 %) | 8,605 → 8,398 (−2.4 %) | +| g++ | `rfft_f64_512` | 1,852 → 1,759 (−5.0 %) | 1,965 → 1,879 (−4.4 %) | +| clang++ | `rfft_f32_512` | 1,252 → 1,283 (+2.5 %) | 1,396 → 1,288 (−7.8 %) | +| clang++ | `rfft_f32_2048` | 6,136 → 6,279 (+2.3 %) | 6,753 → 6,339 (−6.1 %) | +| clang++ | `rfft_f64_512` | 1,516 → 1,404 (−7.4 %) | 1,556 → 1,449 (−6.8 %) | + +Nothing is slower by more than 5 % (the clang++ float forward, +2.3 … ++2.5 %, is the only cell that rose); faster by 5 % or more: the g++ double +forward and every clang++ double and float-inverse cell. A different +machine from #42's table above, so the two tables do not compare cell by +cell. + +**A finding for the class, not the engine.** On Hexagon a third of every +floating scenario is `basic_real_fft`'s out-of-place copy (`forward()` and +`inverse()` copy the input before transforming in place), which clang turns +into calls of musl's `memcpy`: 17.0 M of `rfft_f32_512`'s count, 16.8 M of +`rfft_f32_2048`'s, 33.7 M of `rfft_f64_512`'s, the same for any engine. +Consumers that transform in place do not pay it; the ratchet does. Left +alone here: it is outside the engine, and the fixed-point class copies the +same way. + ## Provenance and licensing ### Where the code came from @@ -2641,6 +2850,80 @@ records as arithmetically the package's. The access statement is the implementer's own report. The maintainer's judgement (`NOTICE.md`): since #42 DspTap ships no code derived from the package. Not legal advice. +### The Hexagon tuning pass, clean-room (2026-09-28) + +After #42 merged, MuTap's Hexagon ratchet (clang 19, v68 + HVX) measured the +srdif engine 12–13 % above the port on its chain workloads. No DspTap key +measured Hexagon, so "must not regress" had never been checked there. The +maintainer chose to tune srdif before the consumer took it. The procedure +followed #42's: + +1. **What the orchestrator did, which had read the port.** It added the + `hexagon` bench key (toolchain file, `icount.py` target, `bench.yml` job). + It measured the port's Hexagon counts once, from a detached checkout of + `7a58ebe` built with the same harness, then deleted that checkout. It also + deleted every port build product it had made while diagnosing on Hexagon: + a public-API microbenchmark's binary and its disassembly. What crossed to + the implementer, all numbers, verbatim from the brief: + - the port's scenario counts: 49,733,404 / 54,835,187 / 103,240,010 for + `rfft_f32_512` / `rfft_f32_2048` / `rfft_f64_512`; + - "Per transform pair (forward + inverse) at N = 512 on Hexagon float, a + public-API microbenchmark read about 14.4 k instructions for the + removed engine against 18.3 k for srdif; double about 31.4 k against + 35.0 k"; + - "at N = 64 srdif was already 1–2 % below", a second measurement of the + port, at another size, that locates the gap. + + The brief also stated that the ratchet counts instructions, not packets. + That was wrong, and the implementer showed it (item 4). +2. **The brief.** The implementer was barred from: + - every copy of the port and the package, and git history before `17db855`; + - the main DspTap clone, the MuTap and MuTap-Max trees (their submodules + carry the port), the other worktrees, and the orchestrator's scratch + except the brief, the conventions file, and #42's accuracy harness and + targets sheet (`clean-engine/TARGETS.md`, `clean-engine/targets/**`, + which use only the public API and were allowed at #42 too); + - `NOTICE.md` and the audit doc; + - every section of this note except the contract, fixed-point, + fp-contraction, Stage 4, size/count and srdif sections; + - every tap/DspTap and tap/MuTap pull request and review, and every other + FFT library's source. + + The brief listed hypotheses to measure, all about the compiler, not about + the port: clang ignoring srdif's GCC pragma, inlining boundaries, index + arithmetic, the complex-multiply form, and the cost of the permutation. +3. **Access statement (the implementer's report).** It opened nothing on the + list. The one exception: broad `grep`s over this file printed single lines + from sections it could not read. These were: + - every heading; + - three history lines; + - line 971 of the transliteration-rules history, which names the port's + routines (`makect`, the `bitrv2*` family, `cftfsub` / `cftbsub`, the + `cft*` leaves); + - a line of the bit-identity record naming `bitrv2` / `bitrv2conj` and the + port's 512-point leaf; + - lines of the provenance and library-comparison sections describing + srdif's own routines and a 16-point leaf's operation count. + + These were names and one-line prose, not code. It reported using none of + them. Its history operations were a `git show --stat` of `28befbd` and a + `git archive` of it for the baseline builds. +4. **Result.** "Hexagon (clang) tuning" in "The floating engine (srdif)" + records the changes, what was tried and the counts: + - forced inlining under clang; + - post-pass pairs loaded before either is stored; + - blocks of 16 and fewer run one level at a time in memory; + - one fused-pass group per loop step for double; + - one multiply-subtract form for float. + + Every Hexagon floating scenario ends 3–8 % below the port, and every + Cortex-M float key 0.06–1.1 % below #42's baselines. No output bit moved + at `-ffp-contract=off`. + + The implementer found that the `hexagon` key counts packets (up to four + instructions each), not instructions, so it rewards packing as well as + instruction count. + ### Comparison with other FFT libraries (2026-09-27) Review B compared the srdif engine with Ooura's package. A separate diff --git a/include/tap/dsp/fft/srdif.h b/include/tap/dsp/fft/srdif.h index 887ac3d..c8905ea 100644 --- a/include/tap/dsp/fft/srdif.h +++ b/include/tap/dsp/fft/srdif.h @@ -66,7 +66,7 @@ // tables.h does for the fixed-point tables. // // The arrangement of these pieces (the fused two-level pass, the twiddle -// pairing, the compile-time blocks and register leaves, the table layouts) +// pairing, the compile-time blocks and their in-memory levels, the table layouts) // is derived in the docstrings below. #pragma once @@ -93,6 +93,38 @@ #define TAP_DSP_SRDIF_NOINLINE #endif +// Under clang, the kernel's building blocks (butterflies, groups, the +// compile-time blocks and their levels, the post-pass pairs, the permutation's +// swaps) are forced inline into the pass that calls them (a performance +// attribute only; the arithmetic and its order are the same either way). +// clang's inliner, left to itself, kept them out of line on Hexagon (clang +// 19, v68): the register leaves this engine had then went through a stack +// array and every group through a call. Forcing them was the largest single +// step of the Hexagon tuning (rfft_f32_512 57.66 -> 49.67 M packets, the +// unit the hexagon key counts, with the leaves of the time; the tree as it +// stands reads 54.13 M without the attribute and 45.94 M with it: +// docs/fft-design.md, "Hexagon (clang) tuning"). The gate is every clang, +// not Hexagon alone (AppleClang, clang-cl, IntelLLVM and clang-based Arm +// toolchains take it too); it was tuned on Hexagon, and the review of the +// tuning measured one more clang target: the bench's Cortex-M33 harness +// compiled by clang 18 (thumbv8m.main, -O3) reads 7.1 % / 5.9 % fewer +// instructions at N = 512 / 2048 forced than not. GCC is left to its own +// choices: forced there, the Cortex-M keys with an FPU (M4F, M33, M55 +// without CMSIS) cost 4.5 - 9.1 % more instructions (spills in the larger +// bodies; the soft-float M4 0.1 - 0.2 % less). Not forced when optimizing +// for size (-Os / -Oz define __OPTIMIZE_SIZE__; clang-cl defines it under +// /O1 and /Os): forced, the Hexagon float size probe's .text grows from +// 238,628 to 263,204 bytes. The choice is made per translation unit, so a +// program that mixes size- and speed-optimized translation units gets +// whichever copy of the out-of-line members the linker keeps: "not forced +// under -Os" is a property of a whole-program -Os build, not a guarantee +// per call site. +#if defined(__clang__) && !defined(__OPTIMIZE_SIZE__) +#define TAP_DSP_SRDIF_INLINE __attribute__((always_inline)) +#else +#define TAP_DSP_SRDIF_INLINE +#endif + namespace tap::dsp::detail { /// The engine's trigonometry: cos and sin of 2 pi k / 2^L in integer @@ -339,8 +371,9 @@ namespace tap::dsp::detail { /// 48 of each. Every block of l >= 32 is processed this way and /// recurses on the five blocks the two levels leave: l/4 (the first /// half's half), l/8 twice (its quarters) and l/4 twice (the - /// quarters). Groups run two per loop step in the run-time pass - /// (l >= 128) so its eight row pointers serve both. + /// quarters). In float, groups run two per loop step in the + /// run-time pass (l >= 128) so its eight row pointers serve both; in + /// double one (k_one_group_per_step). /// - Special butterflies, as Sorensen, Heideman and Burrus count them: /// j = 0 (W = 1, additions only) and j = l/8 (W^j = (1 + i)/sqrt 2, /// W^3j = (-1 + i)/sqrt 2: two additions and two multiplications per @@ -355,9 +388,9 @@ namespace tap::dsp::detail { /// walks it backwards (group_high). /// - Compile-time blocks: blocks of 32 and 64 (block<32>, block<64>) /// run with every offset a constant and their own small tables, and - /// blocks of 16 and fewer are register leaves (leaf: loaded, - /// transformed through every remaining level and stored once). The - /// run-time recursion (kernel) handles l >= 128 only. + /// blocks of 16 and fewer run one split-radix level at a time in + /// memory (level, one butterfly at a time) down to blocks of two. + /// The run-time recursion (kernel) handles l >= 128 only. /// /// The tables: one allocation, sized once and built in place by the /// constructor, never touched by a transform except to read. In Samples, @@ -438,11 +471,16 @@ namespace tap::dsp::detail { /// Speed: the instruction-count ratchet's float scenarios (forward plus /// scaled inverse over 2^20 samples, harness included; arm-none-eabi-gcc /// 13.2.1 -O3, QEMU 8.2.2), against the engine it replaced, N = 512 / - /// 2048: Cortex-M4 soft-float -2.71 % / -2.23 %, M4F -4.70 % / -4.49 %, - /// M33 -7.89 % / -7.53 %, M55 without CMSIS -5.98 % / -5.04 % - /// (bench/README.md, docs/fft-design.md, "The floating engine (srdif)"). - /// On x86-64 (a same-machine A/B of the targets sheet's hostbench, g++ - /// 13.3, pinned core, 11 runs; the predecessor's medians are review A's + /// 2048: Cortex-M4 soft-float -2.78 % / -2.28 %, M4F -5.76 % / -5.51 %, + /// M33 -8.24 % / -7.98 %, M55 without CMSIS -6.33 % / -5.52 % (at #42, + /// before the Hexagon tuning: -2.71 / -2.23, -4.70 / -4.49, -7.89 / + /// -7.53, -5.98 / -5.04 %); Hexagon (clang 19.1.5 -O3, -mv68 -mhvx, + /// qemu-hexagon 8.2.2, which counts packets) float -7.64 % / -5.88 %, + /// double at N = 512 -3.20 %, where #42 read +15.9 % / +19.8 % and + /// +7.2 % (bench/README.md; docs/fft-design.md, "The floating engine + /// (srdif)" and its "Hexagon (clang) tuning"). On x86-64 (a same-machine + /// A/B of the targets sheet's hostbench, g++ 13.3, pinned core, 11 runs, + /// at #42, before the Hexagon tuning; the predecessor's medians are review A's /// of tap/DspTap#42, this engine's measured the same way on the same /// machine, within 2 % of review A's own re-run of the previous tree), /// N = 256 … 4096, forward / inverse against the engine it replaced: @@ -659,7 +697,7 @@ namespace tap::dsp::detail { } } - static void swap2(Sample* p, Sample* q) noexcept { + TAP_DSP_SRDIF_INLINE static void swap2(Sample* p, Sample* q) noexcept { const Sample p0 = p[0]; const Sample p1 = p[1]; const Sample q0 = q[0]; @@ -681,12 +719,42 @@ namespace tap::dsp::detail { Sample i; }; - static cv ld(const Sample* p) noexcept { return {p[0], p[1]}; } - static void st(Sample* p, cv v) noexcept { + TAP_DSP_SRDIF_INLINE static cv ld(const Sample* p) noexcept { return {p[0], p[1]}; } + TAP_DSP_SRDIF_INLINE static void st(Sample* p, cv v) noexcept { p[0] = v.r; p[1] = v.i; } + /// a b - c d, the first product rounded in a statement of its own. + /// + /// Without contraction, and with FLT_EVAL_METHOD == 0, this is the + /// same two products and one subtraction, rounded the same way, as + /// the plain expression (the output bits and the fingerprints do not + /// move; the fingerprint test asserts FLT_EVAL_METHOD == 0). Under + /// excess precision (x87, FLT_EVAL_METHOD == 2) the split statement + /// rounds ab to float where the plain expression would not. The float spelling + /// is for the compilers that contract within a statement (clang, + /// AppleClang): on `a * b - c * d` clang fuses the left product and + /// negates the right one, fma(a, b, -(c d)), which costs Hexagon a + /// separate negation in the two slots the float arithmetic itself + /// needs; here the statement is `ab - c * d`, which it fuses as + /// fma(-c, d, ab), one Hexagon multiply-subtract (Rx -= sfmpy(Rs, + /// Rt)). Hexagon float scenarios -0.65 % / -0.43 % packets (N = 512 / 2048). + /// GCC contracts across statements after SSA and compiles both + /// spellings the same (the Cortex-M counts are identical). Double + /// keeps the plain expression: Hexagon has no double fused + /// multiply-add, and the split spelling there only moved the schedule + /// (rfft_f64_512 +0.14 %). + TAP_DSP_SRDIF_INLINE static Sample mul_sub(Sample a, Sample b, Sample c, Sample d) noexcept { + if constexpr (std::is_same_v) { + const Sample ab = a * b; + return ab - c * d; + } + else { + return a * b - c * d; + } + } + /// The three kinds of split-radix butterfly, by twiddle. enum class tw { one, eighth, general }; @@ -698,7 +766,7 @@ namespace tap::dsp::detail { /// eighth: j = l/8, W^j = (1 + i)/sqrt 2, W^3j = (-1 + i)/sqrt 2; /// general: w1 = W^j, w3 = W^3j as given. template - static void bf(cv& x0, cv& x1, cv& x2, cv& x3, cv w1 = {}, cv w3 = {}) noexcept { + TAP_DSP_SRDIF_INLINE static void bf(cv& x0, cv& x1, cv& x2, cv& x3, cv w1 = {}, cv w3 = {}) noexcept { const Sample t1r = x0.r - x2.r; const Sample t1i = x0.i - x2.i; const Sample t2r = x1.r - x3.r; @@ -743,12 +811,12 @@ namespace tap::dsp::detail { } else { if constexpr (Inverse) { - x2 = {ur * w1.r + ui * w1.i, ui * w1.r - ur * w1.i}; - x3 = {vr * w3.r + vi * w3.i, vi * w3.r - vr * w3.i}; + x2 = {ur * w1.r + ui * w1.i, mul_sub(ui, w1.r, ur, w1.i)}; + x3 = {vr * w3.r + vi * w3.i, mul_sub(vi, w3.r, vr, w3.i)}; } else { - x2 = {ur * w1.r - ui * w1.i, ur * w1.i + ui * w1.r}; - x3 = {vr * w3.r - vi * w3.i, vr * w3.i + vi * w3.r}; + x2 = {mul_sub(ur, w1.r, ui, w1.i), ur * w1.i + ui * w1.r}; + x3 = {mul_sub(vr, w3.r, vi, w3.i), vr * w3.i + vi * w3.r}; } } } @@ -759,7 +827,8 @@ namespace tap::dsp::detail { /// level-l/2 butterfly at j of the first half (c1, c3), whose four /// inputs are exactly the first-half outputs of the other two. template - static void group(Sample* p, std::size_t e, cv a1, cv a3, cv b1, cv b3, cv c1, cv c3) noexcept { + TAP_DSP_SRDIF_INLINE static void group(Sample* p, std::size_t e, cv a1, cv a3, cv b1, cv b3, cv c1, + cv c3) noexcept { const std::size_t d = 2 * e; cv x0 = ld(p); cv x2 = ld(p + 2 * d); @@ -784,21 +853,21 @@ namespace tap::dsp::detail { /// The four kinds of group in a fused pass over a block of length l = 8e. template - static void group_first(Sample* a, std::size_t e) noexcept { + TAP_DSP_SRDIF_INLINE static void group_first(Sample* a, std::size_t e) noexcept { // j = 0: W = 1, then the eighth-turn at j + e, then W = 1 at level l/2. group(a, e, {}, {}, {}, {}, {}, {}); } /// j in [1, e/2), twelve twiddles at w. template - static void group_low(Sample* a, std::size_t e, const Sample* w) noexcept { + TAP_DSP_SRDIF_INLINE static void group_low(Sample* a, std::size_t e, const Sample* w) noexcept { group(a, e, {w[0], w[1]}, {w[2], w[3]}, {w[4], w[5]}, {w[6], w[7]}, {w[8], w[9]}, {w[10], w[11]}); } /// j = e/2 = l/16: W_16^1, W_16^3; W_16^3, W_16^9 = -W_16^1; the eighth-turn at level l/2. template - static void group_mid(Sample* a, std::size_t e) noexcept { + TAP_DSP_SRDIF_INLINE static void group_mid(Sample* a, std::size_t e) noexcept { const cv c16{k_cos_pi8, k_sin_pi8}; const cv s16{k_sin_pi8, k_cos_pi8}; group(a, e, c16, s16, s16, {-k_cos_pi8, -k_sin_pi8}, {}, {}); @@ -808,11 +877,20 @@ namespace tap::dsp::detail { /// W^3j = -i conj W^3(j'+e), W^(j+e) = i conj W^j', W^3(j+e) = -i conj W^3j', /// and at level l/2 W^j = i conj W^j', W^3j = -i conj W^3j'. template - static void group_high(Sample* a, std::size_t e, const Sample* w) noexcept { + TAP_DSP_SRDIF_INLINE static void group_high(Sample* a, std::size_t e, const Sample* w) noexcept { group(a, e, {w[5], w[4]}, {-w[7], -w[6]}, {w[1], w[0]}, {-w[3], -w[2]}, {w[9], w[8]}, {-w[11], -w[10]}); } + /// Whether the run-time fused pass runs one group per loop step + /// (double) or two (float, whose eight row pointers then serve both + /// groups at constant offsets). Two groups of doubles hold 32 values + /// and 24 twiddles, twice Hexagon's register file, and spill; + /// measured on Hexagon, one per step: rfft_f64_512 -0.38 %, the float + /// scenarios +0.48 % / +0.99 % (N = 512 / 2048), so each keeps its + /// best. The Cortex-M keys measure float only. + static constexpr bool k_one_group_per_step = sizeof(Sample) == 8; + /// The fused pass over a block of l >= 128 complex values: level l and /// level l/2 of its first half; the table entry of j is j (m/l) steps. template @@ -821,36 +899,71 @@ namespace tap::dsp::detail { const std::size_t h = e / 2; const std::size_t stride = 12 * ((m_n / 2) / l); // Samples per table step at this length group_first(a, e); - // j = 1 alone, then two groups per step (e/2 - 1 is odd): the - // eight row pointers serve both groups at constant offsets. const Sample* w = m_tables.data() + m_n / 2 + 48 + stride - 12; - group_low(a + 2, e, w); - w += stride; - for (std::size_t j = 2; j < h; j += 2, w += 2 * stride) { - group_low(a + 2 * j, e, w); - group_low(a + 2 * j + 2, e, w + stride); + if constexpr (k_one_group_per_step) { + for (std::size_t j = 1; j < h; ++j, w += stride) { + group_low(a + 2 * j, e, w); + } + group_mid(a + 2 * h, e); + // j in (e/2, e) descending through the table: j' = e - j. + for (std::size_t j = h + 1; j < e; ++j) { + w -= stride; + group_high(a + 2 * j, e, w); + } } - group_mid(a + 2 * h, e); - // j in (e/2, e) descending through the table: j' = e - j. - w -= stride; - group_high(a + 2 * (h + 1), e, w); - for (std::size_t j = h + 2; j < e; j += 2) { - w -= 2 * stride; - group_high(a + 2 * j, e, w + stride); - group_high(a + 2 * j + 2, e, w); + else { + // j = 1 alone, then two groups per step (e/2 - 1 is odd): the + // eight row pointers serve both groups at constant offsets. + group_low(a + 2, e, w); + w += stride; + for (std::size_t j = 2; j < h; j += 2, w += 2 * stride) { + group_low(a + 2 * j, e, w); + group_low(a + 2 * j + 2, e, w + stride); + } + group_mid(a + 2 * h, e); + // j in (e/2, e) descending through the table: j' = e - j. + w -= stride; + group_high(a + 2 * (h + 1), e, w); + for (std::size_t j = h + 2; j < e; j += 2) { + w -= 2 * stride; + group_high(a + 2 * j, e, w + stride); + group_high(a + 2 * j + 2, e, w); + } } } /// A block of L <= 64 complex values with everything known at compile - /// time: the fused pass (L = 32, 64) over the small tables, or a leaf. + /// time: the fused pass over the small tables (L = 32, 64), one + /// split-radix level in memory (L = 4, 8, 16) or the final pair. + /// + /// Below 32 every level runs in memory, one butterfly at a time, + /// down to blocks of two; there are no register leaves. Until the + /// Hexagon tuning blocks of 16 and fewer were register leaves (loaded + /// once, transformed through every remaining level in registers, + /// stored once), which on the ratchet's targets costs more than it + /// saves: a leaf of 16 holds 32 values, the M4F's whole float register + /// file and all of Hexagon's general registers (a double takes two), + /// and spills. The arithmetic of every output is the same either way + /// (the output bits do not move). Measured in this tree with register + /// leaves of at most 16 / 8 / 4 values restored, against the pairs + /// (docs/fft-design.md, "Hexagon (clang) tuning"), millions of + /// packets on Hexagon and of instructions on the Cortex-M keys: + /// Hexagon rfft_f32_512 46.10 / 45.88 / 45.89 / 45.94, + /// rfft_f32_2048 51.79 / 51.58 / 51.55 / 51.61, rfft_f64_512 101.30 / + /// 100.63 / 100.12 / 99.94; M4F rfft_f32_512 92.54 / 92.00 / 91.85 / + /// 91.55, rfft_f32_2048 106.30 / 105.83 / 105.58 / 105.15; the M33 and + /// the M55 without CMSIS 0.3 - 0.5 % below leaves of 8 as well, the + /// soft-float M4 flat. Pairs cost Hexagon float 0.1 % against its best + /// (leaves of 4 or 8) and are the best everywhere else, in one code + /// path for both profiles. template - static void block(Sample* a, const Sample* small) noexcept { + TAP_DSP_SRDIF_INLINE static void block(Sample* a, const Sample* small) noexcept { if constexpr (L >= 32) { constexpr std::size_t e = L / 8; constexpr std::size_t h = e / 2; const Sample* const w = small + (L == 32 ? 0 : 12); // this length's table group_first(a, e); - [&](std::index_sequence) { + [&](std::index_sequence) TAP_DSP_SRDIF_INLINE { (group_low(a + 2 * (J + 1), e, w + 12 * J), ...); (group_high(a + 2 * (e - 1 - J), e, w + 12 * J), ...); }(std::make_index_sequence{}); @@ -861,52 +974,57 @@ namespace tap::dsp::detail { block(a + L, small); block(a + L + L / 2, small); } + else if constexpr (L >= 4) { + level(a); + block(a, small); + if constexpr (L >= 8) { // blocks of one value are their own transforms + block(a + L, small); + block(a + L + L / 2, small); + } + } else { - (void)small; // the leaves need no table - leaf(a); + static_assert(L == 2, "blocks of 2 ... 64 values"); + (void)small; // the pairs need no table + const cv x0 = ld(a); + const cv x1 = ld(a + 2); + st(a, {x0.r + x1.r, x0.i + x1.i}); + st(a + 2, {x0.r - x1.r, x0.i - x1.i}); } } - /// The split-radix recursion over a block held in registers (a leaf of - /// 2 ... 16 complex values), block by block. + /// The first split-radix level of a block of L = 4, 8 or 16 values, in + /// memory, one butterfly at a time: j = 0 (W = 1), j = L/8 (the + /// eighth turn) and, at L = 16, j = 1 and 3 (W_16^1, W_16^3; W_16^3, + /// W_16^9 = -W_16^1). template - static void leaf_blocks(cv* v) noexcept { - if constexpr (L == 1) { - (void)v; // a block of one value is its own transform + TAP_DSP_SRDIF_INLINE static void level(Sample* a) noexcept { + [&](std::index_sequence) + TAP_DSP_SRDIF_INLINE { (level_bf(a), ...); }(std::make_index_sequence{}); + } + + template + TAP_DSP_SRDIF_INLINE static void level_bf(Sample* a) noexcept { + constexpr std::size_t q = L / 4; + cv x0 = ld(a + 2 * J); + cv x1 = ld(a + 2 * (J + q)); + cv x2 = ld(a + 2 * (J + 2 * q)); + cv x3 = ld(a + 2 * (J + 3 * q)); + if constexpr (J == 0) { + bf(x0, x1, x2, x3); } - else if constexpr (L == 2) { - const cv x0 = v[0]; - const cv x1 = v[1]; - v[0] = {x0.r + x1.r, x0.i + x1.i}; - v[1] = {x0.r - x1.r, x0.i - x1.i}; + else if constexpr (J == L / 8) { + bf(x0, x1, x2, x3); } - else if constexpr (L >= 4) { - static_assert(L <= 16, "leaves are 2 ... 16 complex values"); - constexpr std::size_t q = L / 4; - bf(v[0], v[q], v[2 * q], v[3 * q]); - if constexpr (L >= 8) { - constexpr std::size_t e = L / 8; - bf(v[e], v[q + e], v[2 * q + e], v[3 * q + e]); - } - if constexpr (L == 16) { // j = 1: W16^1, W16^3; j = 3: W16^3, W16^9 = -W16^1 - const cv c16{k_cos_pi8, k_sin_pi8}; - const cv s16{k_sin_pi8, k_cos_pi8}; - bf(v[1], v[5], v[9], v[13], c16, s16); - bf(v[3], v[7], v[11], v[15], s16, {-k_cos_pi8, -k_sin_pi8}); - } - leaf_blocks(v + L / 2); - leaf_blocks(v + 3 * q); - leaf_blocks(v); + else if constexpr (J == 1) { // L = 16: W_16^1, W_16^3 + bf(x0, x1, x2, x3, {k_cos_pi8, k_sin_pi8}, {k_sin_pi8, k_cos_pi8}); } - } - - template - static void leaf(Sample* a) noexcept { - [&](std::index_sequence) { - cv v[L] = {ld(a + 2 * I)...}; - leaf_blocks(v); - (st(a + 2 * I, v[I]), ...); - }(std::make_index_sequence{}); + else { // L = 16, J = 3: W_16^3, W_16^9 = -W_16^1 + bf(x0, x1, x2, x3, {k_sin_pi8, k_cos_pi8}, {-k_cos_pi8, -k_sin_pi8}); + } + st(a + 2 * J, x0); + st(a + 2 * (J + q), x1); + st(a + 2 * (J + 2 * q), x2); + st(a + 2 * (J + 3 * q), x3); } /// Split-radix DIF over l complex values at a (bit-reversed output). @@ -970,38 +1088,77 @@ namespace tap::dsp::detail { Sample* const end = a + m; post_pair(pk, pj, c); for (pk += 2, pj -= 2, c += 2; pk < end; pk += 4, pj -= 4, c += 4) { - post_pair(pk, pj, c); - post_pair(pk + 2, pj - 2, c + 2); + post_two(pk, pj, c); } } + /// The four outputs of one bin pair of the post-pass. + struct post_out { + Sample kr; + Sample ki; + Sample jr; + Sample ji; + }; + /// One bin pair of the post-pass: u = a[k], v = conj a[m - k], /// G = C_k (u - v) (conj C_k for the inverse), a[k] <- u - G, - /// a[m - k] <- conj(v + G). + /// a[m - k] <- conj(v + G); u, v and C_k given, the new a[k] and + /// conj a[m - k] returned. template - static void post_pair(Sample* pk, Sample* pj, const Sample* c) noexcept { - const Sample ur = pk[0]; - const Sample ui = pk[1]; - const Sample vr = pj[0]; - const Sample vi = pj[1]; + TAP_DSP_SRDIF_INLINE static post_out post_math(Sample ur, Sample ui, Sample vr, Sample vi, Sample cr, + Sample ci) noexcept { const Sample dr = ur - vr; const Sample di = ui + vi; - const Sample cr = c[0]; - const Sample ci = c[1]; Sample gr; Sample gi; if constexpr (Inverse) { gr = dr * cr + di * ci; - gi = di * cr - dr * ci; + gi = mul_sub(di, cr, dr, ci); } else { - gr = dr * cr - di * ci; + gr = mul_sub(dr, cr, di, ci); gi = dr * ci + di * cr; } - pk[0] = ur - gr; - pk[1] = ui - gi; - pj[0] = vr + gr; - pj[1] = vi - gi; + return {ur - gr, ui - gi, vr + gr, vi - gi}; + } + + template + TAP_DSP_SRDIF_INLINE static void post_pair(Sample* pk, Sample* pj, const Sample* c) noexcept { + const post_out o = post_math(pk[0], pk[1], pj[0], pj[1], c[0], c[1]); + pk[0] = o.kr; + pk[1] = o.ki; + pj[0] = o.jr; + pj[1] = o.ji; + } + + /// Two bin pairs, (k, m - k) and (k + 1, m - k - 1), every load before + /// any store. The compiler cannot tell a store through pk from a later + /// load through pj, so pair by pair the second pair's loads wait for + /// the first pair's stores, and on Hexagon, which issues up to four + /// instructions per packet and counts packets, the two pairs' arithmetic + /// then cannot share them: without this, the Hexagon scenarios read + /// +3.3 % / +3.0 % (float, N = 512 / 2048) and +1.0 % (double). The + /// Cortex-M keys pay at most 0.02 % for it (the soft-float M4). + template + TAP_DSP_SRDIF_INLINE static void post_two(Sample* pk, Sample* pj, const Sample* c) noexcept { + const Sample u0r = pk[0]; + const Sample u0i = pk[1]; + const Sample u1r = pk[2]; + const Sample u1i = pk[3]; + const Sample v1r = pj[-2]; + const Sample v1i = pj[-1]; + const Sample v0r = pj[0]; + const Sample v0i = pj[1]; + const post_out o0 = post_math(u0r, u0i, v0r, v0i, c[0], c[1]); + const post_out o1 = post_math(u1r, u1i, v1r, v1i, c[2], c[3]); + pk[0] = o0.kr; + pk[1] = o0.ki; + pk[2] = o1.kr; + pk[3] = o1.ki; + pj[-2] = o1.jr; + pj[-1] = o1.ji; + pj[0] = o0.jr; + pj[1] = o0.ji; } std::size_t m_n; diff --git a/scripts/icount.py b/scripts/icount.py index 1a2e1ca..cbd1835 100755 --- a/scripts/icount.py +++ b/scripts/icount.py @@ -8,7 +8,7 @@ instruction-counting plugin (tools/qemu_insn_plugin), then compares against bench/baselines.json. - icount.py --target {m4-softfp,m4f,m33,m55,m55-ooura} --build-dir DIR + icount.py --target {m4-softfp,m4f,m33,m55,m55-ooura,hexagon} --build-dir DIR --plugin LIB [--update] [--record FILE] [--baselines bench/baselines.json] [--tolerance 0.03] icount.py --merge FILE [FILE ...] [--baselines bench/baselines.json] @@ -62,6 +62,10 @@ "m33": "mps2-an505", "m55": "mps3-an547", "m55-ooura": "mps3-an547", + # Not a system model: Hexagon runs hosted (hexagon-unknown-linux-musl, + # static) under qemu-hexagon user-mode emulation, built with plugin + # support (bench.yml; cmake/hexagon-linux-musl.cmake). + "hexagon": None, } PREFIX = "tap_dsp_icount_" DONE_MARKER = "TAP_DSP_ICOUNT_DONE ok=1" @@ -72,9 +76,11 @@ def qemu_cmd(target: str, plugin: str, binary: str) -> list[str]: # "-d plugin" routes qemu_plugin_outs() to stderr; without it the count # line is silently dropped. - machine = MACHINES.get(target) - if machine is None: + if target not in MACHINES: raise SystemExit(f"unknown target {target}") + if target == "hexagon": + return ["qemu-hexagon", "-d", "plugin", "-plugin", plugin, binary] + machine = MACHINES[target] return ["qemu-system-arm", "-M", machine, "-nographic", "-semihosting", "-d", "plugin", "-plugin", plugin, "-kernel", binary] diff --git a/tests/test_fft_routing.cpp b/tests/test_fft_routing.cpp index a3fe3b1..c9dc46b 100644 --- a/tests/test_fft_routing.cpp +++ b/tests/test_fft_routing.cpp @@ -47,14 +47,15 @@ namespace { // Sizes small enough for the QEMU legs (Part 10) and wide enough to cross // every code path the engine dispatches on (fft/srdif.h, "Structure"): - // the kernel of M = N/2 complex values as the register leaves alone - // (M = 2, 4, 8, 16: N = 4, 8, 16, 32), the compile-time blocks (M = 32, - // 64: N = 64, 128) and the run-time fused pass above them (M >= 128: - // N = 256 ... 4096, each fused pass recursing down to the compile-time - // blocks), and the post-pass's single-pair (N = 8) and paired-loop - // (N >= 16) forms. The certified geometries 512 and 2048 and 4096 (the - // largest the emulated legs run) are among them. An engine whose range - // excludes a size (CMSIS: 32 … 4096) is checked at the sizes it supports. + // the kernel of M = N/2 complex values as the small blocks alone (M = 2, + // 4, 8, 16: N = 4, 8, 16, 32; levels in memory down to pairs), the + // compile-time blocks (M = 32, 64: N = 64, 128) and the run-time fused + // pass above them (M >= 128: N = 256 ... 4096, each fused pass recursing + // down to the compile-time blocks), and the post-pass's single-pair + // (N = 8) and paired-loop (N >= 16) forms. The certified geometries 512 + // and 2048 and 4096 (the largest the emulated legs run) are among them. + // An engine whose range excludes a size (CMSIS: 32 … 4096) is checked at + // the sizes it supports. constexpr std::size_t k_sizes[] = {4, 8, 16, 32, 64, 128, 256, 512, 2048, 4096}; template