From 7c361c300788b45544c7d3120f8f542deb1ed8b7 Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Thu, 17 Sep 2026 18:12:20 +0000 Subject: [PATCH 1/3] Stage 1 prerequisites: CI flag typo, stale bare-metal filter, fingerprint harness Three MuTap-side prerequisites for the DspTap FFT plan (DspTap docs/audit-fft-and-code-smells.md, Stage 1; review items P5 and P2). CI flag typo. The Cortex-M55 "Ooura FFT fallback" leg passed -DMUTAP_FFT_CMSIS=OFF; the option is TAP_DSP_FFT_CMSIS (DspTap's CMakeLists.txt, and the name MuTap's own CMakeLists.txt and the M33 toolchain already use). CMake ignored the unknown variable, so that leg rebuilt CMSIS and the Ooura-float32-on-M55 profile had no coverage. The leg now passes the real option, tees the configure log and fails if the "float32 FFT backend = CMSIS" line appears, and asserts the cache value. The same misspelling is corrected in the icount job's comment, docs/optimization.md and bench/README.md; optimization.md's pointers to the FFT wrapper, its tests and the vendored CMSIS subset now name DspTap, where they live, and its vDSP section records that MuTap turns TAP_DSP_FFT_ACCELERATE back off (#31). Stale bare-metal filter. tests/bare_metal_main.cpp and the Hexagon selection in tests/CMakeLists.txt named real_fft_test/*, RealFftCrossPrecision.* and CertifiedGeometries/fft_backend_parity.*, which moved to DspTap with the FFT and no longer exist in mutap_tests. Pruned; every remaining name was checked against the test sources, and the host binary selects 52 tests under the pruned filter (the >= 30 empty-run guard is unchanged). Fingerprint harness. tests/branchless_parity_check.cpp, until now the only bit-exact gate in MuTap, becomes tests/fingerprint_harness.cpp (CMake target mutap_fingerprint, a ctest test on every target): a fixed xorshift-driven synthetic echo corpus, generated from integer state with basic IEEE arithmetic only (no libm, no , no clock, no files) and rounded once to float so both profiles see identical values, runs through fdaf, fd_kalman, pem_afc, the residual suppressor, the learned suppressor and the two preset chains in float and double, printing one "FINGERPRINT " line per pair (FNV-1a over the raw output bytes). Two runs on the same host and compiler are identical; the pin-bump workflow is documented at the top of the file (build, run, bump the pin, rebuild, run, diff). The branchless-parity job compiles the same file twice as before and now diffs every FINGERPRINT line instead of one field, and the hosted matrix legs record the lines in their logs so two CI runs can be diffed. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- .github/workflows/ci.yml | 59 ++++-- bench/README.md | 2 +- docs/optimization.md | 89 +++++---- tests/CMakeLists.txt | 24 ++- tests/bare_metal_main.cpp | 9 +- tests/branchless_parity_check.cpp | 65 ------- tests/fingerprint_harness.cpp | 305 ++++++++++++++++++++++++++++++ 7 files changed, 425 insertions(+), 128 deletions(-) delete mode 100644 tests/branchless_parity_check.cpp create mode 100644 tests/fingerprint_harness.cpp diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a1fa7ed..1506135 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -86,6 +86,12 @@ jobs: - name: Test run: ctest --test-dir build -C Release --output-on-failure + # The bit-identity gate for DspTap pin bumps (tests/fingerprint_harness.cpp) + # already ran as a test above; run it once more verbosely so the log + # carries the FINGERPRINT lines and two CI logs can be diffed. + - name: Fingerprints + run: ctest --test-dir build -C Release -R '^mutap_fingerprint$' -V + # These three rows were the per-process bifurcation in #31. Now that the # alignment fix has removed the draw, repeating them says something # different but still worth recording: whether the remaining failures are @@ -159,21 +165,31 @@ jobs: - name: Build run: cmake --build build -j 4 - # The default leg above builds the CMSIS-DSP Helium FFT backend, which is - # ON by default on the bare-metal M55 profile (docs/optimization.md) — so - # ctest exercises the vendored third_party/cmsis-dsp subset and its - # Ooura-contract reconciliation on the full emulated battery. + # The default leg above builds DspTap's CMSIS-DSP Helium FFT backend, + # which is ON by default on the bare-metal M55 profile (TAP_DSP_FFT_CMSIS, + # docs/optimization.md) — so ctest exercises the vendored + # submodules/dsptap/third_party/cmsis-dsp subset and its Ooura-contract + # reconciliation on the full emulated battery. - name: Test under emulation (CMSIS Helium FFT — default) run: ctest --test-dir build --output-on-failure # Second leg keeps the Ooura float32 fallback alive: same emulated battery - # with the backend forced OFF, so -DMUTAP_FFT_CMSIS=OFF cannot bitrot. + # with DspTap's backend option forced OFF. The option is TAP_DSP_FFT_CMSIS + # (submodules/dsptap/CMakeLists.txt); this leg once passed a misspelled + # -DMUTAP_FFT_CMSIS=OFF, which CMake ignored, so it rebuilt CMSIS and the + # Ooura-float32-on-M55 profile had no coverage at all. The configure log + # is checked for the backend line so that can never silently recur. - name: Configure (Ooura FFT fallback) - run: > - cmake -B build-ooura - -DCMAKE_BUILD_TYPE=MinSizeRel - -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake - -DMUTAP_FFT_CMSIS=OFF + run: | + set -eo pipefail + cmake -B build-ooura \ + -DCMAKE_BUILD_TYPE=MinSizeRel \ + -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake \ + -DTAP_DSP_FFT_CMSIS=OFF | tee /tmp/configure-ooura.log + if grep -q 'float32 FFT backend = CMSIS' /tmp/configure-ooura.log; then + echo "::error::the Ooura fallback leg configured the CMSIS backend"; exit 1 + fi + grep '^TAP_DSP_FFT_CMSIS:BOOL=OFF' build-ooura/CMakeCache.txt - name: Build (Ooura FFT fallback) run: cmake --build build-ooura -j 4 @@ -303,8 +319,9 @@ jobs: # The suppressor's pass-1 estimator has two bit-identical forms selected by # MUTAP_SUPPRESSOR_BRANCHLESS (branch-free on Arm Helium, branchy elsewhere; # see include/mutap/postfilter.h). No single build compiles both, so this job - # compiles tests/branchless_parity_check.cpp once per macro value and diffs - # its output fingerprint — the guard that the two forms cannot drift apart. + # compiles the fingerprint harness (tests/fingerprint_harness.cpp — the same + # binary that gates DspTap pin bumps) once per macro value and diffs every + # FINGERPRINT line — the guard that the two forms cannot drift apart. branchless-parity: name: Suppressor branch-free/branchy parity runs-on: ubuntu-latest @@ -313,7 +330,7 @@ jobs: - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 with: submodules: recursive - - name: Compile both forms and diff the fingerprint + - name: Compile both forms and diff the fingerprints run: | set -e # Ooura is C: compile with gcc so g++ does not name-mangle rdft/rdft_f. @@ -321,14 +338,18 @@ jobs: gcc -O2 -c submodules/dsptap/third_party/ooura/fftsg_float.c -o /tmp/fftsg_float.o for bl in 0 1; do g++ -std=c++20 -O2 -DMUTAP_SUPPRESSOR_BRANCHLESS=$bl -Iinclude -Isubmodules/dsptap/include \ - tests/branchless_parity_check.cpp /tmp/fftsg.o /tmp/fftsg_float.o -o /tmp/parity_$bl + tests/fingerprint_harness.cpp /tmp/fftsg.o /tmp/fftsg_float.o -o /tmp/parity_$bl /tmp/parity_$bl | tee /tmp/out_$bl done - a=$(awk '{print $2}' /tmp/out_0); b=$(awk '{print $2}' /tmp/out_1) - if [ "$a" != "$b" ]; then - echo "::error::branchy ($a) and branch-free ($b) suppressor forms diverged"; exit 1 + # Only the FINGERPRINT lines are compared (the '#' header names the + # form and differs by design); an empty run must not pass either. + grep '^FINGERPRINT ' /tmp/out_0 > /tmp/fp_0 + grep '^FINGERPRINT ' /tmp/out_1 > /tmp/fp_1 + test "$(wc -l < /tmp/fp_0)" -ge 14 + if ! diff /tmp/fp_0 /tmp/fp_1; then + echo "::error::branchy and branch-free suppressor forms diverged (see diff above)"; exit 1 fi - echo "both forms fingerprint $a — bit-identical" + echo "all $(wc -l < /tmp/fp_0) fingerprints identical across the two forms — bit-identical" # Deterministic instruction-count ratchet (bench/README.md). Unlike the # wall-clock bench these counts are noise-free under QEMU's TCG plugin, so @@ -374,7 +395,7 @@ jobs: -o /tmp/libinsncount.so tools/qemu_insn_plugin/insn_count.c # Release (-O2) M55 workloads, matching how baselines are recorded. No - # -DMUTAP_FFT_CMSIS here: the M55 profile defaults it ON, so the ratchet + # -DTAP_DSP_FFT_CMSIS here: the M55 profile defaults it ON, so the ratchet # gates the deployed CMSIS-DSP Helium FFT (bench/baselines.json m55). - name: Build M55 workloads run: > diff --git a/bench/README.md b/bench/README.md index 1847fff..3ea52b6 100644 --- a/bench/README.md +++ b/bench/README.md @@ -160,5 +160,5 @@ The **m55** baselines record the CMSIS-DSP Helium FFT, which is the default on the bare-metal M55 profile (`docs/optimization.md`) — ~42% fewer instructions on every layer than the previous Ooura numbers. The ratchet therefore gates the deployed backend. The Ooura float32 path is still available on the M55 with -`-DMUTAP_FFT_CMSIS=OFF` (kept alive by a dedicated CI leg, not by this ratchet). +`-DTAP_DSP_FFT_CMSIS=OFF` (kept alive by a dedicated CI leg, not by this ratchet). The **hexagon** baselines are unaffected — Hexagon stays on scalar Ooura. diff --git a/docs/optimization.md b/docs/optimization.md index c812cf6..8a29176 100644 --- a/docs/optimization.md +++ b/docs/optimization.md @@ -12,7 +12,7 @@ double runs soft-float and is the desktop golden model only). The real FFT is the single hottest kernel in the chain. Profiling the M55 (`-mcpu=cortex-m55`, GCC 13, Helium/MVE) showed GCC does autovectorize the -vendored Ooura float FFT (`third_party/ooura/fftsg_float.c`) — but not nearly +vendored Ooura float FFT (DspTap's `third_party/ooura/fftsg_float.c`) — but not nearly as well as Arm's hand-tuned CMSIS-DSP kernels. Measured, per forward transform, instructions under QEMU: @@ -56,22 +56,26 @@ epsilon). Two things make that acceptable as the M55 default: So the backend is **default ON for the bare-metal M55 embedded profile** — the deployment target — and OFF everywhere else. The Ooura float32 path remains one -flag away (`-DMUTAP_FFT_CMSIS=OFF`) and is kept alive by a dedicated CI leg. +flag away (`-DTAP_DSP_FFT_CMSIS=OFF`) and is kept alive by a dedicated CI leg. +(That leg once passed a misspelled `-DMUTAP_FFT_CMSIS=OFF`, which CMake ignored, +so it silently rebuilt CMSIS; the job now checks the configure log for the +backend line and fails if CMSIS was selected.) ### How it is wired -`include/mutap/fft.h` routes `basic_real_fft` through CMSIS when -`MUTAP_FFT_CMSIS` is defined; the CMake option defaults ON for the bare-metal -M55 profile (`CMAKE_SYSTEM_NAME=Generic` + arm) and OFF everywhere else, so -desktop, the Max/C-ABI host builds (including Apple Silicon arm64), and Hexagon -are untouched. `double` always keeps Ooura regardless. +The FFT lives in DspTap (`submodules/dsptap`, `tap::dsp`; `include/mutap/fft.h` +is a re-export). Its `fft.h` routes `basic_real_fft` through CMSIS when +`TAP_DSP_FFT_CMSIS` is defined; the CMake option of that name defaults ON for +the bare-metal M55 profile (`CMAKE_SYSTEM_NAME=Generic` + arm) and OFF +everywhere else, so desktop, the Max/C-ABI host builds (including Apple Silicon +arm64), and Hexagon are untouched. `double` always keeps Ooura regardless. The wrapper re-presents CMSIS in **Ooura's exact numeric contract** so nothing downstream changes and every intermediate spectrum matches the Ooura build to float epsilon: - **Sign convention.** CMSIS uses the engineering convention exp(−i2π/N); Ooura (and our documented packed layout) uses exp(+i2π/N). The wrapper - conjugates the imaginary bins on every transform. Verified by + conjugates the imaginary bins on every transform. Verified by DspTap's `real_fft_test/0.SignConventionIsPlusI` running on the CMSIS backend. - **Inverse scaling.** CMSIS's inverse RFFT is 1/N-normalized; Ooura's is unnormalized (the caller applies 2/N). The wrapper scales the CMSIS inverse @@ -79,7 +83,7 @@ float epsilon: trip. Both fold into passes the `_inplace` methods already do (~1% overhead). It is on automatically with the M55 toolchain; force the Ooura path with -`-DMUTAP_FFT_CMSIS=OFF`: +`-DTAP_DSP_FFT_CMSIS=OFF`: ```sh # CMSIS backend (default on M55) @@ -87,24 +91,30 @@ cmake -B build-m55 -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake \ -DCMAKE_BUILD_TYPE=Release # Ooura fallback cmake -B build-m55-ooura -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake \ - -DCMAKE_BUILD_TYPE=Release -DMUTAP_FFT_CMSIS=OFF + -DCMAKE_BUILD_TYPE=Release -DTAP_DSP_FFT_CMSIS=OFF ``` Forcing the option ON on a non-Arm processor is a hard error (Helium/NEON only). ### What is validated -- **`tests/test_fft_backend.cpp`** — asserts the CMSIS forward output matches a - direct Ooura `rdft_f` reference bin-for-bin (<5e-6 relative) and that the - round trip reproduces the input, at both certified sizes (512, 2048). Runs on - the M55-CMSIS leg (in `tests/bare_metal_main.cpp`'s selection). -- **The whole `test_fft.cpp` contract suite** (packing, +i sign convention, - Parseval, float-tracks-double) exercises `basic_real_fft`, so it - re-validates the CMSIS backend automatically when the option is on. +- **DspTap's `tests/test_fft_backend.cpp`** — asserts the CMSIS forward output + matches a direct Ooura `rdft_f` reference bin-for-bin (<5e-6 relative) and + that the round trip reproduces the input, at both certified sizes (512, + 2048). The FFT suites moved to DspTap with the FFT and run in its CI, not in + MuTap's `tests/bare_metal_main.cpp` selection. +- **The whole DspTap `test_fft.cpp` contract suite** (packing, +i sign + convention, Parseval, float-tracks-double) exercises `basic_real_fft`, + so it re-validates the CMSIS backend automatically when the option is on. - **The full emulated float32 ITU battery** runs on the M55 with the CMSIS backend (now the default): 58/58 tests pass — the AEC still meets every asserted float32 gate on CMSIS FFTs. A dedicated CI leg re-runs the same - battery with `-DMUTAP_FFT_CMSIS=OFF` to keep the Ooura fallback honest. + battery with `-DTAP_DSP_FFT_CMSIS=OFF` to keep the Ooura fallback honest. +- **`tests/fingerprint_harness.cpp`** (`mutap_fingerprint`) prints one FNV-1a + fingerprint per (component, profile) over a fixed corpus on every CI leg, + including both M55 legs, so a CMSIS-vs-Ooura or pin-to-pin difference in any + output sample is visible as a diff of two logs. It is the bit-identity gate + every DspTap pin bump runs. ### Hexagon: deferred @@ -116,19 +126,26 @@ shape. Hexagon stays on scalar Ooura until an HVX FFT is available. ### Refreshing the vendored CMSIS subset -`third_party/cmsis-dsp/` is a minimal subset (8 sources + header closure), -pinned by commit in `third_party/cmsis-dsp/VENDOR.md`. To bump it: re-run the -`gcc -M` closure over the eight sources for `-mcpu=cortex-m55`, copy exactly the -files it opens, update `VENDOR.md`, then re-run `tests/test_fft_backend.cpp` and -the full float32 battery on the M55 leg. Do not hand-edit vendored sources. +DspTap's `third_party/cmsis-dsp/` is a minimal subset (8 sources + header +closure), pinned by commit in its `VENDOR.md`. To bump it (in DspTap): re-run +the `gcc -M` closure over the eight sources for `-mcpu=cortex-m55`, copy exactly +the files it opens, update `VENDOR.md`, then re-run DspTap's +`tests/test_fft_backend.cpp` and, after bumping the pin here, the full float32 +battery on the M55 leg. Do not hand-edit vendored sources. ## FFT backend: Apple vDSP on macOS -The same seam carries a second backend: on macOS the float32 real FFT routes -through Apple's **vDSP** (Accelerate) instead of Ooura (`MUTAP_FFT_ACCELERATE`, -default ON for Apple). This is the FFT the `mutap.aec~` Max external runs, and -it is the *only* fast FFT Apple Silicon gets — the CMSIS backend is scoped to -the bare-metal M55, so arm64 macOS was on Ooura before this. +The same seam carries a second backend: on macOS the float32 real FFT can +route through Apple's **vDSP** (Accelerate) instead of Ooura +(`TAP_DSP_FFT_ACCELERATE`, DspTap's default ON for Apple). It is the *only* +fast FFT Apple Silicon gets — the CMSIS backend is scoped to the bare-metal +M55, so arm64 macOS was on Ooura before this. **MuTap now turns it back OFF** +(root `CMakeLists.txt`, tap/MuTap#31): on spectra with exactly-empty bins the +alignment-selected vDSP kernel measured far less accurate than Ooura and the +G.168 tone rows failed on it, and Apple documents the routines as free to +rearrange arithmetic. The numbers below stand as the measurement behind that +decision; `-DTAP_DSP_FFT_ACCELERATE=ON` still selects it for anyone who wants +to re-measure. ### Measured (Apple Silicon, macOS CI runner) @@ -168,13 +185,15 @@ deployment precision *through vDSP itself*, not merely parity against Ooura. ### How it is wired / validated -Default ON for Apple (`APPLE` in CMake), OFF elsewhere, mutually exclusive with -the CMSIS backend; CMake links `-framework Accelerate` and defines the macro on -the `mutap` interface target. Force Ooura with `-DMUTAP_FFT_ACCELERATE=OFF`. -Validation is automatic: the `macos-latest` CI job builds with the backend on by -default, so `tests/test_fft_backend.cpp` (bin-for-bin vs Ooura `rdft_f`), the -whole `test_fft.cpp` contract suite, and the full float32 battery all run on -vDSP there. The vDSP setup's read-only twiddle tables are held by a shared_ptr +In DspTap: default ON for Apple (`APPLE` in CMake), OFF elsewhere, mutually +exclusive with the CMSIS backend; CMake links `-framework Accelerate` and +defines the macro on the `tap::dsp` interface target. MuTap's root +`CMakeLists.txt` sets `TAP_DSP_FFT_ACCELERATE` OFF (not FORCE), so an explicit +`-DTAP_DSP_FFT_ACCELERATE=ON` still wins. DspTap's `tests/test_fft_backend.cpp` +(bin-for-bin vs Ooura `rdft_f`) and `test_fft.cpp` contract suite validate the +backend in DspTap's own macOS CI; MuTap's `macos-latest` job gates the Ooura +float32 path it actually ships and records the backend it built. The vDSP +setup's read-only twiddle tables are held by a shared_ptr so `basic_real_fft` keeps value semantics; transforms are noexcept and allocation-free. diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index af2d493..d3fd652 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -69,13 +69,31 @@ else() # of it the double-typed adaptive suites — target-independent math # already covered on every host platform. Run the same emulation-sized # selection as the Cortex-M55 leg (tests/bare_metal_main.cpp): the - # float32 typed suites, the double FFT, the LP/conditioning suites, the - # float closed-loop scenarios and the validation/contract tests. + # float32 typed suites, the double Kalman core, the LP/conditioning + # suites, the float closed-loop scenarios and the validation/contract + # tests. (The FFT suites left with the FFT: they run in DspTap's own CI.) set(mutap_emulated_selection "") if(CMAKE_CROSSCOMPILING) set(mutap_emulated_selection TEST_FILTER - "real_fft_test/0.*:real_fft_test/1.*:RealFftCrossPrecision.*:CertifiedGeometries/fft_backend_parity.*:fdaf_test/0.*:FdafCrossPrecision.*:FdafConfigValidation.*:FdafRtContract.*:fd_kalman_test/0.*:fd_kalman_test/1.*:kalman_loop_test/0.*:FdKalmanConfigValidation.*:FdKalmanRtContract.*:Levinson.*:LpcPredictor.*:SpeechPredictor.*:WarpedLpcPredictor.*:PredictorConfigValidation.*:pem_afc_test/0.*:PemAfcConfigValidation.*:PemAfcRtContract.*:closed_loop_test/0.*:burst_test/0.*:aec_test/0.*:AdaptationControlConfigValidation.*:nn_suppressor_test/0.*:NnSuppressorCrossPrecision.*:NnChainFloat32.*") + "fdaf_test/0.*:FdafCrossPrecision.*:FdafConfigValidation.*:FdafRtContract.*:fd_kalman_test/0.*:fd_kalman_test/1.*:kalman_loop_test/0.*:FdKalmanConfigValidation.*:FdKalmanRtContract.*:Levinson.*:LpcPredictor.*:SpeechPredictor.*:WarpedLpcPredictor.*:PredictorConfigValidation.*:pem_afc_test/0.*:PemAfcConfigValidation.*:PemAfcRtContract.*:closed_loop_test/0.*:burst_test/0.*:aec_test/0.*:AdaptationControlConfigValidation.*:nn_suppressor_test/0.*:NnSuppressorCrossPrecision.*:NnChainFloat32.*") endif() gtest_discover_tests(mutap_tests DISCOVERY_TIMEOUT 120 ${mutap_emulated_selection} PROPERTIES TIMEOUT 900) endif() + +# The bit-identity gate for every DspTap pin bump (tests/fingerprint_harness.cpp): +# one FNV-1a fingerprint per (component, profile) over a fixed synthetic corpus. +# A plain executable with its own main — no gtest — so the branchless-parity +# CI job can also compile the same file standalone. Registered as a test on +# every target so each CI log carries the lines; on bare metal the pass +# criterion is the completion line, as for mutap_tests_emulated. +add_executable(mutap_fingerprint fingerprint_harness.cpp) +target_link_libraries(mutap_fingerprint PRIVATE + MuTap::MuTap + mutap_warnings) +add_test(NAME mutap_fingerprint COMMAND mutap_fingerprint) +set_tests_properties(mutap_fingerprint PROPERTIES TIMEOUT 900) +if(MUTAP_BARE_METAL) + set_tests_properties(mutap_fingerprint PROPERTIES + PASS_REGULAR_EXPRESSION "FINGERPRINT_COMPLETE") +endif() diff --git a/tests/bare_metal_main.cpp b/tests/bare_metal_main.cpp index 40c5017..b7d05e1 100644 --- a/tests/bare_metal_main.cpp +++ b/tests/bare_metal_main.cpp @@ -2,7 +2,8 @@ // qemu-system-arm): there is no argv on the target, so the // emulation-appropriate selection is baked in. This is a POSITIVE filter: // the float32 typed suites (the embedded profile this target exists for), -// the double FFT (small; exercises the soft-float path), the LP / +// the double Kalman core (exercises the soft-float path; the FFT suites +// moved to DspTap with the FFT and run in its own CI), the LP / // conditioning suite, the float closed-loop scenarios including the PEM // canceller's tonal headline, the float-tracks-double oracle check, and // the learned suppressor's float profile with its own oracle check (the @@ -29,8 +30,6 @@ int main() { // Typed-suite naming: /0 = float, /1 = double (sample_types order). ::testing::GTEST_FLAG(filter) = - "real_fft_test/0.*:real_fft_test/1.*:RealFftCrossPrecision.*:" - "CertifiedGeometries/fft_backend_parity.*:" "fdaf_test/0.*:FdafCrossPrecision.*:FdafConfigValidation.*:FdafRtContract.*:" "fd_kalman_test/0.*:fd_kalman_test/1.*:FdKalmanConfigValidation.*:FdKalmanRtContract.*:" "Levinson.*:LpcPredictor.*:SpeechPredictor.*:WarpedLpcPredictor.*:PredictorConfigValidation.*:" @@ -44,8 +43,8 @@ int main() { // A filter typo selects zero tests and RUN_ALL_TESTS() returns 0 — an // empty run must not pass green. Checked after the run because gtest // only applies the filter inside RUN_ALL_TESTS. The on-target selection - // is ~57 tests; 30 leaves headroom for legitimate removals without - // masking a typo. + // is 52 tests (the FFT suites left with the FFT); 30 leaves headroom + // for legitimate removals without masking a typo. const int selected = ::testing::UnitTest::GetInstance()->test_to_run_count(); if (selected < 30) { std::printf("only %d tests selected (expected >= 30): filter is broken\n", selected); diff --git a/tests/branchless_parity_check.cpp b/tests/branchless_parity_check.cpp deleted file mode 100644 index a9e35c3..0000000 --- a/tests/branchless_parity_check.cpp +++ /dev/null @@ -1,65 +0,0 @@ -// SPDX-License-Identifier: MIT -// Copyright 2026 MuTap contributors -// -// Bit-identity guard for the suppressor's two pass-1 forms (see -// MUTAP_SUPPRESSOR_BRANCHLESS in include/mutap/postfilter.h). The branch-free -// form is compiled on Arm Helium and the branchy form everywhere else, so no -// single build exercises both; CI compiles THIS file twice — once per macro -// value — and diffs the fingerprint below. Identical fingerprints prove the -// two forms produce sample-exact output, which is what lets the branch-free -// form ride the compliance battery that certified the branchy one. -// -// Standalone (not part of mutap_tests): the whole point is to build it once -// per macro value. See the "branchless parity" step in .github/workflows/ci.yml. -#include -#include -#include -#include - -#include "mutap/postfilter.h" - -int main() { - auto chain = tap::mu::aec_chain(tap::mu::aec_chain_preset(256, 8, 48000.0)); - constexpr int B = 256; - std::vector x(B), y(B), o(B); - - std::uint32_t s = 0x12345u; - auto nx = [&s]() noexcept { - s ^= s << 13; - s ^= s >> 17; - s ^= s << 5; - return static_cast(s) / 2147483648.0f - 1.0f; - }; - - // FNV-1a over the raw bits of every output sample: any single-ULP - // divergence between the two forms flips the fingerprint. - std::uint64_t fp = 1469598103934665603ULL; - auto mix = [&fp](float v) noexcept { - std::uint32_t b; - __builtin_memcpy(&b, &v, 4); - fp = (fp ^ b) * 1099511628211ULL; - }; - - // A corpus varied enough to drive every data-dependent branch the two - // forms disagree on if they ever diverge: level sweeps, double-talk - // bursts (near-end incoherent with the echo), and quiet stretches. - for (int blk = 0; blk < 600; ++blk) { - const float amp = 0.05f + 0.1f * std::fabs(std::sin(blk * 0.03f)); - const bool dt = (blk / 40) % 3 == 0; - for (int i = 0; i < B; ++i) { - const float far = amp * nx(); - const float echo = 0.3f * far; - const float near = dt ? 0.08f * std::sin((blk * B + i) * 0.05f) + 0.03f * nx() : 0.0f; - x[i] = far; - y[i] = echo + near + 0.001f * nx(); - } - chain.process_block(x.data(), y.data(), o.data()); - for (float v : o) { - mix(v); - } - } - - std::printf("MUTAP_SUPPRESSOR_PARITY_FP %016llx branchless=%d\n", static_cast(fp), - MUTAP_SUPPRESSOR_BRANCHLESS); - return 0; -} diff --git a/tests/fingerprint_harness.cpp b/tests/fingerprint_harness.cpp new file mode 100644 index 0000000..613c8c1 --- /dev/null +++ b/tests/fingerprint_harness.cpp @@ -0,0 +1,305 @@ +// SPDX-License-Identifier: MIT +// Copyright 2026 MuTap contributors +// +// THE BIT-IDENTITY GATE FOR EVERY DspTap PIN BUMP. +// +// Runs a fixed, xorshift-driven synthetic echo corpus through each of MuTap's +// signal-processing components in BOTH numeric profiles and prints one line +// per (component, profile): +// +// FINGERPRINT <16 hex digits> +// +// where the digits are FNV-1a (64-bit) over the raw bytes of every output +// sample the component produced. Any single-ULP change anywhere in a +// component's output stream flips its line. Components: +// +// fdaf partitioned_fdaf, the NLMS core with its control stack +// fd_kalman partitioned_fdkf at the certified preset calibration +// pem_afc the FDAF-PEM-AFROW canceller (speech predictor + fdaf) +// postfilter residual_suppressor alone, fed a fixed (e, yhat) pair +// nn_suppressor the learned post-filter alone, deterministic weights +// aec_chain the certified chain: fd_kalman + postfilter (preset) +// aec_chain_nn the learned chain: fd_kalman + nn_suppressor (preset) +// +// The corpus is generated in double from integer state with basic IEEE +// arithmetic only (no libm, no , no wall clock, no filesystem) and +// rounded once to float, so both profiles consume identical sample values +// and the harness runs unchanged on bare metal. Determinism holds per +// (host, compiler, flags): libm and fp-contraction differ across hosts, so +// two fingerprints are comparable only when both runs were produced by the +// same build configuration on the same machine — which is exactly the pin- +// bump workflow below. +// +// How to diff two DspTap pins (the check every submodule bump runs): +// +// cmake -S . -B build -DCMAKE_BUILD_TYPE=Release +// cmake --build build --target mutap_fingerprint +// ./build/tests/mutap_fingerprint > /tmp/before.txt +// git -C submodules/dsptap checkout +// cmake --build build --target mutap_fingerprint +// ./build/tests/mutap_fingerprint > /tmp/after.txt +// diff /tmp/before.txt /tmp/after.txt +// +// An empty diff is the proof that the bump changed no output sample; every +// stage of the FFT plan (DspTap docs/audit-fft-and-code-smells.md) states +// which lines here must be unchanged and which are expected to move (the +// float profile through an FFT port; never the double golden model unless +// the plan says so). CI runs this binary on every hosted leg (the +// "Fingerprints" step records the lines in the log, so two CI logs can be +// diffed the same way) and once more, compiled twice, as the suppressor's +// branch-free/branchy parity check (MUTAP_SUPPRESSOR_BRANCHLESS, see +// include/mutap/postfilter.h): the two builds must print identical lines. +// +// Builds two ways: as the normal CMake target mutap_fingerprint (a ctest +// test on every hosted target), and standalone as the parity job compiles +// it (g++ on this file plus DspTap's Ooura .c files — see the +// branchless-parity job in .github/workflows/ci.yml). +#include +#include +#include +#include +#include + +#include "mutap/fd_kalman.h" +#include "mutap/fdaf.h" +#include "mutap/nn_chain.h" +#include "mutap/pem_afc.h" +#include "mutap/postfilter.h" + +namespace { + + // The certified reference geometry: block 256, 8 partitions (2048 taps) + // at 48 kHz — the calibration point of every preset. + constexpr std::size_t k_block = 256; + constexpr std::size_t k_partitions = 8; + constexpr double k_sample_rate = 48000.0; + // Corpus length in blocks (~2.1 s of audio). Long enough to carry every + // canceller through convergence and into the double-talk / quiet + // segments; short enough for the double profile under soft-float + // emulation on the Cortex-M legs. + constexpr std::size_t k_blocks = 400; + + /// xorshift32 -> uniform in [-1, 1), exactly representable in float: + /// a 24-bit signed integer times a power of two, so the corpus is the + /// same value set in both profiles by construction. + class xorshift32 { + public: + explicit xorshift32(std::uint32_t seed) noexcept + : m_s(seed) {} + + double next() noexcept { + m_s ^= m_s << 13; + m_s ^= m_s >> 17; + m_s ^= m_s << 5; + const auto q = static_cast(m_s >> 8) - 8388608; // [-2^23, 2^23) + return static_cast(q) * (1.0 / 8388608.0); + } + + private: + std::uint32_t m_s; + }; + + /// FNV-1a over raw sample bytes. Any ULP flips it. + class fingerprint { + public: + template + void mix(const Sample* v, std::size_t n) noexcept { + unsigned char bytes[sizeof(Sample)]; + for (std::size_t i = 0; i < n; ++i) { + std::memcpy(bytes, &v[i], sizeof(Sample)); + for (const unsigned char b : bytes) { + m_h = (m_h ^ b) * 1099511628211ULL; + } + } + } + + unsigned long long value() const noexcept { return static_cast(m_h); } + + private: + std::uint64_t m_h = 1469598103934665603ULL; + }; + + /// One block of the corpus, produced on demand from fixed integer state + /// (no stored corpus: the bare-metal targets are RAM-bound). Every + /// component gets a fresh source, so every component sees the same + /// blocks in the same order. + /// + /// x far end: uniform noise under a slow triangular level sweep + /// y mic: 3-tap sparse echo of x + near end + a tiny floor + /// e a converged canceller's residual (y minus 95 % of the echo) + /// yhat that canceller's echo estimate (the echo itself) + /// + /// The near end is present in every third 40-block stretch (double + /// talk): one-pole-colored noise plus a white component, incoherent + /// with the echo — enough structure to drive every data-dependent path + /// the suppressors and the control stacks branch on. + template + class corpus_source { + public: + corpus_source() + : m_far(0x12345u) + , m_near(0x9E3779B9u) + , m_floor(0x2545F491u) + , m_history(k_history, 0.0) + , x(k_block) + , y(k_block) + , e(k_block) + , yhat(k_block) {} + + void generate(std::size_t blk) noexcept { + const std::size_t p = blk % 200; + const double tri = static_cast(p < 100 ? p : 200 - p) * 0.01; + const double amp = 0.05 + 0.1 * tri; + const bool dt = (blk / 40) % 3 == 0; + for (std::size_t i = 0; i < k_block; ++i) { + const double far = amp * m_far.next(); + // Delay line: newest sample at index 0. + for (std::size_t k = k_history - 1; k > 0; --k) { + m_history[k] = m_history[k - 1]; + } + m_history[0] = far; + const double echo = + 0.25 * m_history[k_block / 2] - 0.12 * m_history[k_block] + 0.06 * m_history[3 * k_block / 2]; + double near = 0.0; + if (dt) { + m_colored = 0.9 * m_colored + 0.1 * m_near.next(); + near = 0.6 * m_colored + 0.03 * m_near.next(); + } + const double floor = 0.001 * m_floor.next(); + // Round once to float; the double profile widens exactly. + x[i] = static_cast(static_cast(far)); + y[i] = static_cast(static_cast(echo + near + floor)); + e[i] = static_cast(static_cast(0.05 * echo + near + floor)); + yhat[i] = static_cast(static_cast(echo)); + } + } + + private: + static constexpr std::size_t k_history = 3 * k_block / 2 + 1; + xorshift32 m_far; + xorshift32 m_near; + xorshift32 m_floor; + std::vector m_history; + double m_colored = 0.0; + + public: + std::vector x, y, e, yhat; + }; + + /// Deterministic live weights at the shipping 48 kHz geometry (the + /// values do not matter, only that the network's gains vary with the + /// input and never change between runs). + tap::mu::nn_suppressor_weights nn_weights() { + const tap::mu::nn_geometry g{48000.0, 256, 26, 64, 96}; + xorshift32 rng(0xC0FFEE11u); + auto fill = [&rng](std::vector& v, std::size_t n) { + v.resize(n); + for (auto& w : v) { + w = static_cast(rng.next() * 0.3); + } + }; + tap::mu::nn_suppressor_weights w; + w.geometry = g; + fill(w.dense_in_w, g.dense * g.features()); + fill(w.dense_in_b, g.dense); + fill(w.gru_w_ih, 3 * g.gru * g.dense); + fill(w.gru_w_hh, 3 * g.gru * g.gru); + fill(w.gru_b_ih, 3 * g.gru); + fill(w.gru_b_hh, 3 * g.gru); + fill(w.dense_out_w, g.bands * g.gru); + fill(w.dense_out_b, g.bands); + return w; + } + + template + const char* profile_name() noexcept { + if constexpr (sizeof(Sample) == sizeof(float)) { + return "float"; + } + else { + return "double"; + } + } + + /// Runs `step(source, out)` once per block and prints the component's + /// line. Selection of the component's inputs is the step's business. + template + void run(const char* component, Step&& step) { + corpus_source src; + std::vector out(k_block); + fingerprint fp; + for (std::size_t blk = 0; blk < k_blocks; ++blk) { + src.generate(blk); + step(src, out.data()); + fp.mix(out.data(), k_block); + } + std::printf("FINGERPRINT %s %s %016llx\n", component, profile_name(), fp.value()); + } + + template + void run_profile() { + using namespace tap::mu; + using src_t = corpus_source; + + { + typename partitioned_fdaf::config cfg; + cfg.block_size = k_block; + cfg.partitions = k_partitions; + cfg.ipc_step_scaling = true; + cfg.ipc_freeze_threshold = Sample(0.1); + cfg.transient_freeze_ratio = Sample(8); + partitioned_fdaf f(cfg); + run("fdaf", [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + } + { + partitioned_fdkf f(aec_chain_preset(k_block, k_partitions, k_sample_rate).canceller); + run("fd_kalman", + [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + } + { + typename pem_afc::config cfg; + cfg.fdaf.block_size = k_block; + cfg.fdaf.partitions = k_partitions; + pem_afc f(cfg); + run("pem_afc", [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + } + { + auto cfg = aec_chain_preset(k_block, k_partitions, k_sample_rate).postfilter; + cfg.block_size = k_block; + residual_suppressor f(cfg); + run("postfilter", + [&f](const src_t& s, Sample* out) { f.process_block(s.e.data(), s.yhat.data(), out); }); + } + { + auto cfg = aec_chain_nn_preset(k_block, k_partitions, k_sample_rate, nn_weights()).postfilter; + nn_suppressor f(std::move(cfg)); + run("nn_suppressor", + [&f](const src_t& s, Sample* out) { f.process_block(s.e.data(), s.yhat.data(), out); }); + } + { + aec_chain f(aec_chain_preset(k_block, k_partitions, k_sample_rate)); + run("aec_chain", + [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + } + { + aec_chain_nn f(aec_chain_nn_preset(k_block, k_partitions, k_sample_rate, nn_weights())); + run("aec_chain_nn", + [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + } + } + +} // namespace + +int main() { + // Informational header (not part of the diffed lines): the geometry and + // the suppressor form this binary compiled, so a log is self-describing. + std::printf("# mutap_fingerprint block=%u partitions=%u rate=%u blocks=%u branchless=%d\n", + static_cast(k_block), static_cast(k_partitions), + static_cast(k_sample_rate), static_cast(k_blocks), MUTAP_SUPPRESSOR_BRANCHLESS); + run_profile(); + run_profile(); + // CTest's pass criterion on bare metal, where semihosting does not + // reliably propagate the exit code: printed only after every line. + std::printf("FINGERPRINT_COMPLETE\n"); + return 0; +} From 72e7e69389a4d61e04fc217412e2babf8bf92987 Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Thu, 17 Sep 2026 18:38:08 +0000 Subject: [PATCH 2/3] CI: record the on-target fingerprints on the emulated legs ctest hides a passing test's output, so the M55 (CMSIS and Ooura), M33 and Hexagon jobs re-run mutap_fingerprint verbosely after their battery: each log now carries the FINGERPRINT lines for that target and backend, so the two M55 backends and any two pins can be diffed from the logs. The hosted matrix already had this step. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- .github/workflows/ci.yml | 15 +++++++++++++++ tests/fingerprint_harness.cpp | 7 ++++--- 2 files changed, 19 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 1506135..563d07e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -173,6 +173,12 @@ jobs: - name: Test under emulation (CMSIS Helium FFT — default) run: ctest --test-dir build --output-on-failure + # ctest hides a passing test's output, so re-run the fingerprint harness + # verbosely: the log then carries the on-target FINGERPRINT lines for + # this backend, diffable against the Ooura leg below and across pins. + - name: Fingerprints (CMSIS Helium FFT) + run: ctest --test-dir build -R '^mutap_fingerprint$' -V + # Second leg keeps the Ooura float32 fallback alive: same emulated battery # with DspTap's backend option forced OFF. The option is TAP_DSP_FFT_CMSIS # (submodules/dsptap/CMakeLists.txt); this leg once passed a misspelled @@ -197,6 +203,9 @@ jobs: - name: Test under emulation (Ooura FFT fallback) run: ctest --test-dir build-ooura --output-on-failure + - name: Fingerprints (Ooura FFT fallback) + run: ctest --test-dir build-ooura -R '^mutap_fingerprint$' -V + # Cortex-M33 (Raspberry Pi Pico 2 W / RP2350 class: single-precision FPU, # no FP64, no MVE) on QEMU's MPS2+ AN505 model — the wake-word plan's named # embedded target. Same Armv8-M startup as the M55 leg, Ooura float32 FFT @@ -230,6 +239,9 @@ jobs: - name: Test under emulation run: ctest --test-dir build --output-on-failure + - name: Fingerprints + run: ctest --test-dir build -R '^mutap_fingerprint$' -V + # Cross-compile for Qualcomm Hexagon (hexagon-unknown-linux-musl, HVX # auto-vectorization on) and run the FULL test suite under qemu-hexagon # user-mode emulation: the third target of the one-core/three-targets @@ -292,6 +304,9 @@ jobs: - name: Test under emulation run: ctest --test-dir build --output-on-failure + - name: Fingerprints + run: ctest --test-dir build -R '^mutap_fingerprint$' -V + # Keeps the benchmarks compiling and runnable; never a performance gate # (shared runners are noise — see bench/README.md). bench-smoke: diff --git a/tests/fingerprint_harness.cpp b/tests/fingerprint_harness.cpp index 613c8c1..83d466b 100644 --- a/tests/fingerprint_harness.cpp +++ b/tests/fingerprint_harness.cpp @@ -44,9 +44,10 @@ // stage of the FFT plan (DspTap docs/audit-fft-and-code-smells.md) states // which lines here must be unchanged and which are expected to move (the // float profile through an FFT port; never the double golden model unless -// the plan says so). CI runs this binary on every hosted leg (the -// "Fingerprints" step records the lines in the log, so two CI logs can be -// diffed the same way) and once more, compiled twice, as the suppressor's +// the plan says so). CI runs this binary on every leg, hosted and emulated +// (a "Fingerprints" step records the lines in each log, so two CI logs can +// be diffed the same way, and the M55's CMSIS and Ooura legs against each +// other) and once more, compiled twice, as the suppressor's // branch-free/branchy parity check (MUTAP_SUPPRESSOR_BRANCHLESS, see // include/mutap/postfilter.h): the two builds must print identical lines. // From b2ca86c79bbcae0b0558d8c5be9ea58b2755762d Mon Sep 17 00:00:00 2001 From: Timothy Place Date: Thu, 17 Sep 2026 20:02:56 +0000 Subject: [PATCH 3/3] Review fixes for #50: hash the estimate stream, positive backend assert, docs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hostile reviews A (correctness) and B (process) on tap/MuTap#50. Harness (A#3, A#6, A#7). fdaf and fd_kalman now run the four-argument process_block and both output channels are hashed — the error block and the echo-estimate block — because error = desired - estimate can absorb a one-ULP move of the estimate under a loud near end. All 14 fingerprints change as a consequence (expected; the PR body records the new values). The '#' header prints backend=cmsis|vdsp|ooura from the TAP_DSP_FFT_CMSIS / TAP_DSP_FFT_ACCELERATE compile definitions the FFT header keys on, so every log is self-describing. Double talk now starts at block 40 so each canceller converges in single talk first, as the comment always said. Style: one declaration per line, corpus buffers private with accessors (NL.16 order), std::is_same_v for the profile name, value() returns std::uint64_t (cast at the printf), included. CI (A#4, A#5, B#S4). The Ooura-fallback leg keeps the :BOOL=OFF cache assertion and additionally greps backend=ooura from the harness output of the binary that ran on target — a positive proof of the compiled backend, not an absence check; the CMSIS leg asserts backend=cmsis, M33 and Hexagon backend=ooura. The four emulated legs run the battery with -E '^mutap_fingerprint$' and the harness once, verbosely, in the Fingerprints step (saves ~4 min M33, ~3 min Hexagon, ~1.5 min M55 per run). The parity job runs under set -eo pipefail — GitHub's default run shell is `bash -e {0}` without pipefail, so the pipes through tee would otherwise mask a harness crash — and floors both files at 14 lines; a comment marks it as the Stage 2c (P18) rewrite site. Docs (A#1, A#2, B#S1, B#S2, B#S3). optimization.md: the CMSIS bin-for-bin parity gate is executed by no CI at the current DspTap pin (DspTap's M55 leg is compile-only with tests off) and will be by DspTap's own M55 leg once DspTap #17 lands; the M55 float32 battery is 52/52, not 58/58; the suppressor section names tests/fingerprint_harness.cpp, 400 blocks and 14 lines instead of the deleted file and a 600-block corpus; the flag history is a missed rename at the DspTap extraction (14116f0), not a typo, in ci.yml and the doc alike. HANDOFF note 6 and the PR template's "Submodule pin moved" bullet each point a pin bumper at the harness header's diff procedure. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_019ZPTzNxo5Fe4EtpXXKf7Sy --- .github/pull_request_template.md | 5 +- .github/workflows/ci.yml | 61 ++++++++++----- HANDOFF.md | 5 +- docs/optimization.md | 53 ++++++++----- tests/fingerprint_harness.cpp | 123 ++++++++++++++++++++----------- 5 files changed, 165 insertions(+), 82 deletions(-) diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 8723f79..5179437 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -24,7 +24,10 @@ breaking for every consumer; say which ones. - **Submodule pin moved.** Which pin, and why. After this merges by rebase/squash, repoint any open consumer pins at the identical tree on - `main` so they stay reachable once the branch is deleted. + `main` so they stay reachable once the branch is deleted. For a + `submodules/dsptap` bump, paste the `mutap_fingerprint` diff (before/after + the pin; procedure at the top of `tests/fingerprint_harness.cpp`) and say + which `FINGERPRINT` lines the stage expects to move. - **Notebooks re-executed**, because behavior changed. They are committed executed. - **Style configs** came from a `taphouse` sync, not a hand-edit. A diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 563d07e..8a20906 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -170,21 +170,31 @@ jobs: # docs/optimization.md) — so ctest exercises the vendored # submodules/dsptap/third_party/cmsis-dsp subset and its Ooura-contract # reconciliation on the full emulated battery. + # The battery leaves the fingerprint harness out (-E): ctest hides a + # passing test's output, so the harness runs once, verbosely, in the + # next step instead, and the log carries the on-target FINGERPRINT + # lines for this backend, diffable against the Ooura leg below and + # across pins. The harness prints the backend it was compiled with; + # asserting it is the positive proof of which backend ran. - name: Test under emulation (CMSIS Helium FFT — default) - run: ctest --test-dir build --output-on-failure + run: ctest --test-dir build --output-on-failure -E '^mutap_fingerprint$' - # ctest hides a passing test's output, so re-run the fingerprint harness - # verbosely: the log then carries the on-target FINGERPRINT lines for - # this backend, diffable against the Ooura leg below and across pins. - name: Fingerprints (CMSIS Helium FFT) - run: ctest --test-dir build -R '^mutap_fingerprint$' -V + run: | + set -eo pipefail + ctest --test-dir build -R '^mutap_fingerprint$' -V | tee /tmp/fp-cmsis.log + grep -q 'backend=cmsis' /tmp/fp-cmsis.log # Second leg keeps the Ooura float32 fallback alive: same emulated battery - # with DspTap's backend option forced OFF. The option is TAP_DSP_FFT_CMSIS - # (submodules/dsptap/CMakeLists.txt); this leg once passed a misspelled - # -DMUTAP_FFT_CMSIS=OFF, which CMake ignored, so it rebuilt CMSIS and the - # Ooura-float32-on-M55 profile had no coverage at all. The configure log - # is checked for the backend line so that can never silently recur. + # with DspTap's backend option forced OFF. The option was renamed from + # MUTAP_FFT_CMSIS to TAP_DSP_FFT_CMSIS when the FFT moved to DspTap + # (14116f0); this leg kept the old name, which CMake ignored, so from that + # commit until #50 it rebuilt CMSIS and the Ooura-float32-on-M55 profile + # had no coverage. Guards: the configure log must not select CMSIS, the + # cache entry must be the typed :BOOL=OFF (an unknown -D lands as + # :UNINITIALIZED, so a future rename fails here instead of silently + # rebuilding CMSIS), and the Fingerprints step below asserts the backend + # the binary that ran was actually compiled with. - name: Configure (Ooura FFT fallback) run: | set -eo pipefail @@ -201,10 +211,13 @@ jobs: run: cmake --build build-ooura -j 4 - name: Test under emulation (Ooura FFT fallback) - run: ctest --test-dir build-ooura --output-on-failure + run: ctest --test-dir build-ooura --output-on-failure -E '^mutap_fingerprint$' - name: Fingerprints (Ooura FFT fallback) - run: ctest --test-dir build-ooura -R '^mutap_fingerprint$' -V + run: | + set -eo pipefail + ctest --test-dir build-ooura -R '^mutap_fingerprint$' -V | tee /tmp/fp-ooura.log + grep -q 'backend=ooura' /tmp/fp-ooura.log # Cortex-M33 (Raspberry Pi Pico 2 W / RP2350 class: single-precision FPU, # no FP64, no MVE) on QEMU's MPS2+ AN505 model — the wake-word plan's named @@ -237,10 +250,14 @@ jobs: run: cmake --build build -j 4 - name: Test under emulation - run: ctest --test-dir build --output-on-failure + run: ctest --test-dir build --output-on-failure -E '^mutap_fingerprint$' + # The toolchain file pins CMSIS off (no MVE); the harness must say so. - name: Fingerprints - run: ctest --test-dir build -R '^mutap_fingerprint$' -V + run: | + set -eo pipefail + ctest --test-dir build -R '^mutap_fingerprint$' -V | tee /tmp/fp-m33.log + grep -q 'backend=ooura' /tmp/fp-m33.log # Cross-compile for Qualcomm Hexagon (hexagon-unknown-linux-musl, HVX # auto-vectorization on) and run the FULL test suite under qemu-hexagon @@ -302,10 +319,13 @@ jobs: run: cmake --build build -j 4 - name: Test under emulation - run: ctest --test-dir build --output-on-failure + run: ctest --test-dir build --output-on-failure -E '^mutap_fingerprint$' - name: Fingerprints - run: ctest --test-dir build -R '^mutap_fingerprint$' -V + run: | + set -eo pipefail + ctest --test-dir build -R '^mutap_fingerprint$' -V | tee /tmp/fp-hexagon.log + grep -q 'backend=ooura' /tmp/fp-hexagon.log # Keeps the benchmarks compiling and runnable; never a performance gate # (shared runners are noise — see bench/README.md). @@ -347,8 +367,12 @@ jobs: submodules: recursive - name: Compile both forms and diff the fingerprints run: | - set -e - # Ooura is C: compile with gcc so g++ does not name-mangle rdft/rdft_f. + # GitHub's default run shell is `bash -e {0}` without pipefail, so + # the pipes through tee below would mask a harness crash without it. + set -eo pipefail + # Compiles DspTap's C by path; Stage 2c (P18) rewrites this job when + # the C goes. Ooura is C: compile with gcc so g++ does not + # name-mangle rdft/rdft_f. gcc -O2 -c submodules/dsptap/third_party/ooura/fftsg.c -o /tmp/fftsg.o gcc -O2 -c submodules/dsptap/third_party/ooura/fftsg_float.c -o /tmp/fftsg_float.o for bl in 0 1; do @@ -361,6 +385,7 @@ jobs: grep '^FINGERPRINT ' /tmp/out_0 > /tmp/fp_0 grep '^FINGERPRINT ' /tmp/out_1 > /tmp/fp_1 test "$(wc -l < /tmp/fp_0)" -ge 14 + test "$(wc -l < /tmp/fp_1)" -ge 14 if ! diff /tmp/fp_0 /tmp/fp_1; then echo "::error::branchy and branch-free suppressor forms diverged (see diff above)"; exit 1 fi diff --git a/HANDOFF.md b/HANDOFF.md index 8c11410..0091bb7 100644 --- a/HANDOFF.md +++ b/HANDOFF.md @@ -111,7 +111,10 @@ carries the measured numbers; this is the map: merge orphans the branch SHAs. After any MuTap PR merges, re-pin `MuTap-Max/submodules/MuTap` to the new main tip before (or as part of) the next MuTap-Max merge — a dangling gitlink breaks recursive - clones once branches are cleaned up. + clones once branches are cleaned up. Every `submodules/dsptap` bump + pastes the `mutap_fingerprint` diff (before/after the pin; procedure + at the top of `tests/fingerprint_harness.cpp`) into its PR and says + which `FINGERPRINT` lines the stage expects to move. 7. **min-api pin gotcha** (MuTap-Max): the pinned min-api has no scalar `send()` path on queue-backed outlets — send a pre-allocated `atoms` lvalue (see `m_ipc_atoms` in the external). diff --git a/docs/optimization.md b/docs/optimization.md index 8a29176..08c1083 100644 --- a/docs/optimization.md +++ b/docs/optimization.md @@ -57,9 +57,12 @@ epsilon). Two things make that acceptable as the M55 default: So the backend is **default ON for the bare-metal M55 embedded profile** — the deployment target — and OFF everywhere else. The Ooura float32 path remains one flag away (`-DTAP_DSP_FFT_CMSIS=OFF`) and is kept alive by a dedicated CI leg. -(That leg once passed a misspelled `-DMUTAP_FFT_CMSIS=OFF`, which CMake ignored, -so it silently rebuilt CMSIS; the job now checks the configure log for the -backend line and fails if CMSIS was selected.) +(History: the option was renamed from `MUTAP_FFT_CMSIS` to `TAP_DSP_FFT_CMSIS` +when the FFT moved to DspTap, commit `14116f0`; the leg kept the old name, which +CMake ignored, so from that commit until tap/MuTap#50 it silently rebuilt CMSIS. +The job now asserts the typed cache entry `TAP_DSP_FFT_CMSIS:BOOL=OFF` — an +unknown `-D` lands as `:UNINITIALIZED`, so a rename fails the leg — and the +harness binary that ran prints `backend=ooura`, which the leg greps for.) ### How it is wired @@ -98,23 +101,33 @@ Forcing the option ON on a non-Arm processor is a hard error (Helium/NEON only). ### What is validated -- **DspTap's `tests/test_fft_backend.cpp`** — asserts the CMSIS forward output - matches a direct Ooura `rdft_f` reference bin-for-bin (<5e-6 relative) and - that the round trip reproduces the input, at both certified sizes (512, - 2048). The FFT suites moved to DspTap with the FFT and run in its CI, not in - MuTap's `tests/bare_metal_main.cpp` selection. +- **DspTap's `tests/test_fft_backend.cpp`** (`CertifiedGeometries/ + fft_backend_parity`) asserts the CMSIS forward output matches a direct Ooura + `rdft_f` reference bin-for-bin (<5e-6 relative) and that the round trip + reproduces the input, at both certified sizes (512, 2048). Honest status: at + the current DspTap pin this gate on the CMSIS backend is executed by **no + CI** — the suites moved to DspTap with the FFT (so they are gone from + MuTap's `tests/bare_metal_main.cpp` selection), and DspTap's own M55 leg is + compile-only with its tests off; it will run once DspTap's embedded legs land + (DspTap #17, plan Part 10). Until then the float32 battery below is the only + CMSIS gate anywhere. - **The whole DspTap `test_fft.cpp` contract suite** (packing, +i sign convention, Parseval, float-tracks-double) exercises `basic_real_fft`, - so it re-validates the CMSIS backend automatically when the option is on. -- **The full emulated float32 ITU battery** runs on the M55 with the CMSIS - backend (now the default): 58/58 tests pass — the AEC still meets every - asserted float32 gate on CMSIS FFTs. A dedicated CI leg re-runs the same - battery with `-DTAP_DSP_FFT_CMSIS=OFF` to keep the Ooura fallback honest. + so it re-validates the CMSIS backend automatically wherever it runs with the + option on (same status as above for the M55). +- **The emulated float32 battery** (`mutap_tests_emulated`, the 52-test + selection in `tests/bare_metal_main.cpp`) runs on the M55 with the CMSIS + backend (the default): 52/52 pass — the AEC still meets every asserted + float32 gate on CMSIS FFTs. A dedicated CI leg re-runs the same battery with + `-DTAP_DSP_FFT_CMSIS=OFF` to keep the Ooura fallback honest. - **`tests/fingerprint_harness.cpp`** (`mutap_fingerprint`) prints one FNV-1a fingerprint per (component, profile) over a fixed corpus on every CI leg, - including both M55 legs, so a CMSIS-vs-Ooura or pin-to-pin difference in any - output sample is visible as a diff of two logs. It is the bit-identity gate - every DspTap pin bump runs. + including both M55 legs, so a pin-to-pin difference in any output sample is + visible as a diff of two logs; it is the bit-identity gate every DspTap pin + bump runs (procedure at the top of the file). Between the two M55 legs it + shows what the contract predicts — the seven `double` lines identical, the + seven `float` lines all different — which documents the backends' difference + but asserts nothing about CMSIS accuracy; that is the parity gate's job. ### Hexagon: deferred @@ -220,6 +233,8 @@ Measured (icount, vs the branchy form on each target): So each target runs its faster form; neither regresses. Because it is per-target, no single build compiles both shapes — the `branchless-parity` CI -job compiles `tests/branchless_parity_check.cpp` once per macro value and diffs -an output fingerprint over a 600-block double-talk corpus, guaranteeing the two -forms stay sample-exact. The m55 baselines record the branch-free form. +job compiles the fingerprint harness (`tests/fingerprint_harness.cpp`, the +`mutap_fingerprint` target; see "What is validated" above) once per macro value +and diffs every `FINGERPRINT` line — 14 of them, seven components in both +profiles over the 400-block corpus — guaranteeing the two forms stay +sample-exact. The m55 baselines record the branch-free form. diff --git a/tests/fingerprint_harness.cpp b/tests/fingerprint_harness.cpp index 83d466b..3e05b88 100644 --- a/tests/fingerprint_harness.cpp +++ b/tests/fingerprint_harness.cpp @@ -21,6 +21,11 @@ // aec_chain the certified chain: fd_kalman + postfilter (preset) // aec_chain_nn the learned chain: fd_kalman + nn_suppressor (preset) // +// For fdaf and fd_kalman both output channels are hashed: the error block +// and the echo-estimate block (the four-argument process_block), because +// error = desired - estimate can absorb a one-ULP move of the estimate when +// the near end is loud; the estimate stream alone exposes it. +// // The corpus is generated in double from integer state with basic IEEE // arithmetic only (no libm, no , no wall clock, no filesystem) and // rounded once to float, so both profiles consume identical sample values @@ -28,7 +33,9 @@ // (host, compiler, flags): libm and fp-contraction differ across hosts, so // two fingerprints are comparable only when both runs were produced by the // same build configuration on the same machine — which is exactly the pin- -// bump workflow below. +// bump workflow below. The '#' header names the float32 FFT backend the +// binary was compiled with (backend=cmsis|vdsp|ooura), so a log is +// self-describing and the M55 legs can assert which backend they ran. // // How to diff two DspTap pins (the check every submodule bump runs): // @@ -52,13 +59,15 @@ // include/mutap/postfilter.h): the two builds must print identical lines. // // Builds two ways: as the normal CMake target mutap_fingerprint (a ctest -// test on every hosted target), and standalone as the parity job compiles -// it (g++ on this file plus DspTap's Ooura .c files — see the -// branchless-parity job in .github/workflows/ci.yml). +// test on every target), and standalone as the parity job compiles it (g++ +// on this file plus DspTap's Ooura .c files — see the branchless-parity job +// in .github/workflows/ci.yml). #include #include #include #include +#include +#include #include #include "mutap/fd_kalman.h" @@ -80,6 +89,16 @@ namespace { // emulation on the Cortex-M legs. constexpr std::size_t k_blocks = 400; + // The float32 FFT backend this binary was compiled against, from the + // compile definitions DspTap's fft.h keys on (not the CMake cache). +#if defined(TAP_DSP_FFT_CMSIS) + constexpr const char* k_backend = "cmsis"; +#elif defined(TAP_DSP_FFT_ACCELERATE) + constexpr const char* k_backend = "vdsp"; +#else + constexpr const char* k_backend = "ooura"; +#endif + /// xorshift32 -> uniform in [-1, 1), exactly representable in float: /// a 24-bit signed integer times a power of two, so the corpus is the /// same value set in both profiles by construction. @@ -114,7 +133,7 @@ namespace { } } - unsigned long long value() const noexcept { return static_cast(m_h); } + std::uint64_t value() const noexcept { return m_h; } private: std::uint64_t m_h = 1469598103934665603ULL; @@ -130,10 +149,11 @@ namespace { /// e a converged canceller's residual (y minus 95 % of the echo) /// yhat that canceller's echo estimate (the echo itself) /// - /// The near end is present in every third 40-block stretch (double - /// talk): one-pole-colored noise plus a white component, incoherent - /// with the echo — enough structure to drive every data-dependent path - /// the suppressors and the control stacks branch on. + /// The near end is present in every third 40-block stretch from block + /// 40 on (double talk), so each canceller first converges in single + /// talk: one-pole-colored noise plus a white component, incoherent with + /// the echo — enough structure to drive every data-dependent path the + /// suppressors and the control stacks branch on. template class corpus_source { public: @@ -142,16 +162,16 @@ namespace { , m_near(0x9E3779B9u) , m_floor(0x2545F491u) , m_history(k_history, 0.0) - , x(k_block) - , y(k_block) - , e(k_block) - , yhat(k_block) {} + , m_x(k_block) + , m_y(k_block) + , m_e(k_block) + , m_yhat(k_block) {} void generate(std::size_t blk) noexcept { const std::size_t p = blk % 200; const double tri = static_cast(p < 100 ? p : 200 - p) * 0.01; const double amp = 0.05 + 0.1 * tri; - const bool dt = (blk / 40) % 3 == 0; + const bool dt = (blk / 40) % 3 == 1; for (std::size_t i = 0; i < k_block; ++i) { const double far = amp * m_far.next(); // Delay line: newest sample at index 0. @@ -168,13 +188,18 @@ namespace { } const double floor = 0.001 * m_floor.next(); // Round once to float; the double profile widens exactly. - x[i] = static_cast(static_cast(far)); - y[i] = static_cast(static_cast(echo + near + floor)); - e[i] = static_cast(static_cast(0.05 * echo + near + floor)); - yhat[i] = static_cast(static_cast(echo)); + m_x[i] = static_cast(static_cast(far)); + m_y[i] = static_cast(static_cast(echo + near + floor)); + m_e[i] = static_cast(static_cast(0.05 * echo + near + floor)); + m_yhat[i] = static_cast(static_cast(echo)); } } + const Sample* x() const noexcept { return m_x.data(); } + const Sample* y() const noexcept { return m_y.data(); } + const Sample* e() const noexcept { return m_e.data(); } + const Sample* yhat() const noexcept { return m_yhat.data(); } + private: static constexpr std::size_t k_history = 3 * k_block / 2 + 1; xorshift32 m_far; @@ -182,9 +207,10 @@ namespace { xorshift32 m_floor; std::vector m_history; double m_colored = 0.0; - - public: - std::vector x, y, e, yhat; + std::vector m_x; + std::vector m_y; + std::vector m_e; + std::vector m_yhat; }; /// Deterministic live weights at the shipping 48 kHz geometry (the @@ -214,27 +240,34 @@ namespace { template const char* profile_name() noexcept { - if constexpr (sizeof(Sample) == sizeof(float)) { + if constexpr (std::is_same_v) { return "float"; } else { + static_assert(std::is_same_v, "two profiles: float and double"); return "double"; } } - /// Runs `step(source, out)` once per block and prints the component's - /// line. Selection of the component's inputs is the step's business. + /// Runs `step(source, out, estimate)` once per block and prints the + /// component's line. `out` is always hashed; `estimate` is hashed too + /// when the step writes it (the two cancellers' second output channel). template - void run(const char* component, Step&& step) { + void run(const char* component, bool hashes_estimate, Step&& step) { corpus_source src; std::vector out(k_block); + std::vector estimate(k_block); fingerprint fp; for (std::size_t blk = 0; blk < k_blocks; ++blk) { src.generate(blk); - step(src, out.data()); + step(src, out.data(), estimate.data()); fp.mix(out.data(), k_block); + if (hashes_estimate) { + fp.mix(estimate.data(), k_block); + } } - std::printf("FINGERPRINT %s %s %016llx\n", component, profile_name(), fp.value()); + std::printf("FINGERPRINT %s %s %016llx\n", component, profile_name(), + static_cast(fp.value())); } template @@ -250,53 +283,57 @@ namespace { cfg.ipc_freeze_threshold = Sample(0.1); cfg.transient_freeze_ratio = Sample(8); partitioned_fdaf f(cfg); - run("fdaf", [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + run("fdaf", true, + [&f](const src_t& s, Sample* out, Sample* est) { f.process_block(s.x(), s.y(), out, est); }); } { partitioned_fdkf f(aec_chain_preset(k_block, k_partitions, k_sample_rate).canceller); - run("fd_kalman", - [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + run("fd_kalman", true, + [&f](const src_t& s, Sample* out, Sample* est) { f.process_block(s.x(), s.y(), out, est); }); } { typename pem_afc::config cfg; cfg.fdaf.block_size = k_block; cfg.fdaf.partitions = k_partitions; pem_afc f(cfg); - run("pem_afc", [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + run("pem_afc", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.x(), s.y(), out); }); } { auto cfg = aec_chain_preset(k_block, k_partitions, k_sample_rate).postfilter; cfg.block_size = k_block; residual_suppressor f(cfg); - run("postfilter", - [&f](const src_t& s, Sample* out) { f.process_block(s.e.data(), s.yhat.data(), out); }); + run("postfilter", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.e(), s.yhat(), out); }); } { auto cfg = aec_chain_nn_preset(k_block, k_partitions, k_sample_rate, nn_weights()).postfilter; nn_suppressor f(std::move(cfg)); - run("nn_suppressor", - [&f](const src_t& s, Sample* out) { f.process_block(s.e.data(), s.yhat.data(), out); }); + run("nn_suppressor", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.e(), s.yhat(), out); }); } { aec_chain f(aec_chain_preset(k_block, k_partitions, k_sample_rate)); - run("aec_chain", - [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + run("aec_chain", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.x(), s.y(), out); }); } { aec_chain_nn f(aec_chain_nn_preset(k_block, k_partitions, k_sample_rate, nn_weights())); - run("aec_chain_nn", - [&f](const src_t& s, Sample* out) { f.process_block(s.x.data(), s.y.data(), out); }); + run("aec_chain_nn", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.x(), s.y(), out); }); } } } // namespace int main() { - // Informational header (not part of the diffed lines): the geometry and - // the suppressor form this binary compiled, so a log is self-describing. - std::printf("# mutap_fingerprint block=%u partitions=%u rate=%u blocks=%u branchless=%d\n", + // Informational header (not part of the diffed lines): the geometry, the + // float32 FFT backend and the suppressor form this binary compiled, so a + // log is self-describing and the M55 legs can assert their backend. + std::printf("# mutap_fingerprint block=%u partitions=%u rate=%u blocks=%u backend=%s branchless=%d\n", static_cast(k_block), static_cast(k_partitions), - static_cast(k_sample_rate), static_cast(k_blocks), MUTAP_SUPPRESSOR_BRANCHLESS); + static_cast(k_sample_rate), static_cast(k_blocks), k_backend, + MUTAP_SUPPRESSOR_BRANCHLESS); run_profile(); run_profile(); // CTest's pass criterion on bare metal, where semihosting does not