diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 8723f79..5179437 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -24,7 +24,10 @@ breaking for every consumer; say which ones. - **Submodule pin moved.** Which pin, and why. After this merges by rebase/squash, repoint any open consumer pins at the identical tree on - `main` so they stay reachable once the branch is deleted. + `main` so they stay reachable once the branch is deleted. For a + `submodules/dsptap` bump, paste the `mutap_fingerprint` diff (before/after + the pin; procedure at the top of `tests/fingerprint_harness.cpp`) and say + which `FINGERPRINT` lines the stage expects to move. - **Notebooks re-executed**, because behavior changed. They are committed executed. - **Style configs** came from a `taphouse` sync, not a hand-edit. A diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a1fa7ed..8a20906 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -86,6 +86,12 @@ jobs: - name: Test run: ctest --test-dir build -C Release --output-on-failure + # The bit-identity gate for DspTap pin bumps (tests/fingerprint_harness.cpp) + # already ran as a test above; run it once more verbosely so the log + # carries the FINGERPRINT lines and two CI logs can be diffed. + - name: Fingerprints + run: ctest --test-dir build -C Release -R '^mutap_fingerprint$' -V + # These three rows were the per-process bifurcation in #31. Now that the # alignment fix has removed the draw, repeating them says something # different but still worth recording: whether the remaining failures are @@ -159,27 +165,59 @@ jobs: - name: Build run: cmake --build build -j 4 - # The default leg above builds the CMSIS-DSP Helium FFT backend, which is - # ON by default on the bare-metal M55 profile (docs/optimization.md) — so - # ctest exercises the vendored third_party/cmsis-dsp subset and its - # Ooura-contract reconciliation on the full emulated battery. + # The default leg above builds DspTap's CMSIS-DSP Helium FFT backend, + # which is ON by default on the bare-metal M55 profile (TAP_DSP_FFT_CMSIS, + # docs/optimization.md) — so ctest exercises the vendored + # submodules/dsptap/third_party/cmsis-dsp subset and its Ooura-contract + # reconciliation on the full emulated battery. + # The battery leaves the fingerprint harness out (-E): ctest hides a + # passing test's output, so the harness runs once, verbosely, in the + # next step instead, and the log carries the on-target FINGERPRINT + # lines for this backend, diffable against the Ooura leg below and + # across pins. The harness prints the backend it was compiled with; + # asserting it is the positive proof of which backend ran. - name: Test under emulation (CMSIS Helium FFT — default) - run: ctest --test-dir build --output-on-failure + run: ctest --test-dir build --output-on-failure -E '^mutap_fingerprint$' + + - name: Fingerprints (CMSIS Helium FFT) + run: | + set -eo pipefail + ctest --test-dir build -R '^mutap_fingerprint$' -V | tee /tmp/fp-cmsis.log + grep -q 'backend=cmsis' /tmp/fp-cmsis.log # Second leg keeps the Ooura float32 fallback alive: same emulated battery - # with the backend forced OFF, so -DMUTAP_FFT_CMSIS=OFF cannot bitrot. + # with DspTap's backend option forced OFF. The option was renamed from + # MUTAP_FFT_CMSIS to TAP_DSP_FFT_CMSIS when the FFT moved to DspTap + # (14116f0); this leg kept the old name, which CMake ignored, so from that + # commit until #50 it rebuilt CMSIS and the Ooura-float32-on-M55 profile + # had no coverage. Guards: the configure log must not select CMSIS, the + # cache entry must be the typed :BOOL=OFF (an unknown -D lands as + # :UNINITIALIZED, so a future rename fails here instead of silently + # rebuilding CMSIS), and the Fingerprints step below asserts the backend + # the binary that ran was actually compiled with. - name: Configure (Ooura FFT fallback) - run: > - cmake -B build-ooura - -DCMAKE_BUILD_TYPE=MinSizeRel - -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake - -DMUTAP_FFT_CMSIS=OFF + run: | + set -eo pipefail + cmake -B build-ooura \ + -DCMAKE_BUILD_TYPE=MinSizeRel \ + -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake \ + -DTAP_DSP_FFT_CMSIS=OFF | tee /tmp/configure-ooura.log + if grep -q 'float32 FFT backend = CMSIS' /tmp/configure-ooura.log; then + echo "::error::the Ooura fallback leg configured the CMSIS backend"; exit 1 + fi + grep '^TAP_DSP_FFT_CMSIS:BOOL=OFF' build-ooura/CMakeCache.txt - name: Build (Ooura FFT fallback) run: cmake --build build-ooura -j 4 - name: Test under emulation (Ooura FFT fallback) - run: ctest --test-dir build-ooura --output-on-failure + run: ctest --test-dir build-ooura --output-on-failure -E '^mutap_fingerprint$' + + - name: Fingerprints (Ooura FFT fallback) + run: | + set -eo pipefail + ctest --test-dir build-ooura -R '^mutap_fingerprint$' -V | tee /tmp/fp-ooura.log + grep -q 'backend=ooura' /tmp/fp-ooura.log # Cortex-M33 (Raspberry Pi Pico 2 W / RP2350 class: single-precision FPU, # no FP64, no MVE) on QEMU's MPS2+ AN505 model — the wake-word plan's named @@ -212,7 +250,14 @@ jobs: run: cmake --build build -j 4 - name: Test under emulation - run: ctest --test-dir build --output-on-failure + run: ctest --test-dir build --output-on-failure -E '^mutap_fingerprint$' + + # The toolchain file pins CMSIS off (no MVE); the harness must say so. + - name: Fingerprints + run: | + set -eo pipefail + ctest --test-dir build -R '^mutap_fingerprint$' -V | tee /tmp/fp-m33.log + grep -q 'backend=ooura' /tmp/fp-m33.log # Cross-compile for Qualcomm Hexagon (hexagon-unknown-linux-musl, HVX # auto-vectorization on) and run the FULL test suite under qemu-hexagon @@ -274,7 +319,13 @@ jobs: run: cmake --build build -j 4 - name: Test under emulation - run: ctest --test-dir build --output-on-failure + run: ctest --test-dir build --output-on-failure -E '^mutap_fingerprint$' + + - name: Fingerprints + run: | + set -eo pipefail + ctest --test-dir build -R '^mutap_fingerprint$' -V | tee /tmp/fp-hexagon.log + grep -q 'backend=ooura' /tmp/fp-hexagon.log # Keeps the benchmarks compiling and runnable; never a performance gate # (shared runners are noise — see bench/README.md). @@ -303,8 +354,9 @@ jobs: # The suppressor's pass-1 estimator has two bit-identical forms selected by # MUTAP_SUPPRESSOR_BRANCHLESS (branch-free on Arm Helium, branchy elsewhere; # see include/mutap/postfilter.h). No single build compiles both, so this job - # compiles tests/branchless_parity_check.cpp once per macro value and diffs - # its output fingerprint — the guard that the two forms cannot drift apart. + # compiles the fingerprint harness (tests/fingerprint_harness.cpp — the same + # binary that gates DspTap pin bumps) once per macro value and diffs every + # FINGERPRINT line — the guard that the two forms cannot drift apart. branchless-parity: name: Suppressor branch-free/branchy parity runs-on: ubuntu-latest @@ -313,22 +365,31 @@ jobs: - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 with: submodules: recursive - - name: Compile both forms and diff the fingerprint + - name: Compile both forms and diff the fingerprints run: | - set -e - # Ooura is C: compile with gcc so g++ does not name-mangle rdft/rdft_f. + # GitHub's default run shell is `bash -e {0}` without pipefail, so + # the pipes through tee below would mask a harness crash without it. + set -eo pipefail + # Compiles DspTap's C by path; Stage 2c (P18) rewrites this job when + # the C goes. Ooura is C: compile with gcc so g++ does not + # name-mangle rdft/rdft_f. gcc -O2 -c submodules/dsptap/third_party/ooura/fftsg.c -o /tmp/fftsg.o gcc -O2 -c submodules/dsptap/third_party/ooura/fftsg_float.c -o /tmp/fftsg_float.o for bl in 0 1; do g++ -std=c++20 -O2 -DMUTAP_SUPPRESSOR_BRANCHLESS=$bl -Iinclude -Isubmodules/dsptap/include \ - tests/branchless_parity_check.cpp /tmp/fftsg.o /tmp/fftsg_float.o -o /tmp/parity_$bl + tests/fingerprint_harness.cpp /tmp/fftsg.o /tmp/fftsg_float.o -o /tmp/parity_$bl /tmp/parity_$bl | tee /tmp/out_$bl done - a=$(awk '{print $2}' /tmp/out_0); b=$(awk '{print $2}' /tmp/out_1) - if [ "$a" != "$b" ]; then - echo "::error::branchy ($a) and branch-free ($b) suppressor forms diverged"; exit 1 + # Only the FINGERPRINT lines are compared (the '#' header names the + # form and differs by design); an empty run must not pass either. + grep '^FINGERPRINT ' /tmp/out_0 > /tmp/fp_0 + grep '^FINGERPRINT ' /tmp/out_1 > /tmp/fp_1 + test "$(wc -l < /tmp/fp_0)" -ge 14 + test "$(wc -l < /tmp/fp_1)" -ge 14 + if ! diff /tmp/fp_0 /tmp/fp_1; then + echo "::error::branchy and branch-free suppressor forms diverged (see diff above)"; exit 1 fi - echo "both forms fingerprint $a — bit-identical" + echo "all $(wc -l < /tmp/fp_0) fingerprints identical across the two forms — bit-identical" # Deterministic instruction-count ratchet (bench/README.md). Unlike the # wall-clock bench these counts are noise-free under QEMU's TCG plugin, so @@ -374,7 +435,7 @@ jobs: -o /tmp/libinsncount.so tools/qemu_insn_plugin/insn_count.c # Release (-O2) M55 workloads, matching how baselines are recorded. No - # -DMUTAP_FFT_CMSIS here: the M55 profile defaults it ON, so the ratchet + # -DTAP_DSP_FFT_CMSIS here: the M55 profile defaults it ON, so the ratchet # gates the deployed CMSIS-DSP Helium FFT (bench/baselines.json m55). - name: Build M55 workloads run: > diff --git a/HANDOFF.md b/HANDOFF.md index 8c11410..0091bb7 100644 --- a/HANDOFF.md +++ b/HANDOFF.md @@ -111,7 +111,10 @@ carries the measured numbers; this is the map: merge orphans the branch SHAs. After any MuTap PR merges, re-pin `MuTap-Max/submodules/MuTap` to the new main tip before (or as part of) the next MuTap-Max merge — a dangling gitlink breaks recursive - clones once branches are cleaned up. + clones once branches are cleaned up. Every `submodules/dsptap` bump + pastes the `mutap_fingerprint` diff (before/after the pin; procedure + at the top of `tests/fingerprint_harness.cpp`) into its PR and says + which `FINGERPRINT` lines the stage expects to move. 7. **min-api pin gotcha** (MuTap-Max): the pinned min-api has no scalar `send()` path on queue-backed outlets — send a pre-allocated `atoms` lvalue (see `m_ipc_atoms` in the external). diff --git a/bench/README.md b/bench/README.md index 1847fff..3ea52b6 100644 --- a/bench/README.md +++ b/bench/README.md @@ -160,5 +160,5 @@ The **m55** baselines record the CMSIS-DSP Helium FFT, which is the default on the bare-metal M55 profile (`docs/optimization.md`) — ~42% fewer instructions on every layer than the previous Ooura numbers. The ratchet therefore gates the deployed backend. The Ooura float32 path is still available on the M55 with -`-DMUTAP_FFT_CMSIS=OFF` (kept alive by a dedicated CI leg, not by this ratchet). +`-DTAP_DSP_FFT_CMSIS=OFF` (kept alive by a dedicated CI leg, not by this ratchet). The **hexagon** baselines are unaffected — Hexagon stays on scalar Ooura. diff --git a/docs/optimization.md b/docs/optimization.md index c812cf6..08c1083 100644 --- a/docs/optimization.md +++ b/docs/optimization.md @@ -12,7 +12,7 @@ double runs soft-float and is the desktop golden model only). The real FFT is the single hottest kernel in the chain. Profiling the M55 (`-mcpu=cortex-m55`, GCC 13, Helium/MVE) showed GCC does autovectorize the -vendored Ooura float FFT (`third_party/ooura/fftsg_float.c`) — but not nearly +vendored Ooura float FFT (DspTap's `third_party/ooura/fftsg_float.c`) — but not nearly as well as Arm's hand-tuned CMSIS-DSP kernels. Measured, per forward transform, instructions under QEMU: @@ -56,22 +56,29 @@ epsilon). Two things make that acceptable as the M55 default: So the backend is **default ON for the bare-metal M55 embedded profile** — the deployment target — and OFF everywhere else. The Ooura float32 path remains one -flag away (`-DMUTAP_FFT_CMSIS=OFF`) and is kept alive by a dedicated CI leg. +flag away (`-DTAP_DSP_FFT_CMSIS=OFF`) and is kept alive by a dedicated CI leg. +(History: the option was renamed from `MUTAP_FFT_CMSIS` to `TAP_DSP_FFT_CMSIS` +when the FFT moved to DspTap, commit `14116f0`; the leg kept the old name, which +CMake ignored, so from that commit until tap/MuTap#50 it silently rebuilt CMSIS. +The job now asserts the typed cache entry `TAP_DSP_FFT_CMSIS:BOOL=OFF` — an +unknown `-D` lands as `:UNINITIALIZED`, so a rename fails the leg — and the +harness binary that ran prints `backend=ooura`, which the leg greps for.) ### How it is wired -`include/mutap/fft.h` routes `basic_real_fft` through CMSIS when -`MUTAP_FFT_CMSIS` is defined; the CMake option defaults ON for the bare-metal -M55 profile (`CMAKE_SYSTEM_NAME=Generic` + arm) and OFF everywhere else, so -desktop, the Max/C-ABI host builds (including Apple Silicon arm64), and Hexagon -are untouched. `double` always keeps Ooura regardless. +The FFT lives in DspTap (`submodules/dsptap`, `tap::dsp`; `include/mutap/fft.h` +is a re-export). Its `fft.h` routes `basic_real_fft` through CMSIS when +`TAP_DSP_FFT_CMSIS` is defined; the CMake option of that name defaults ON for +the bare-metal M55 profile (`CMAKE_SYSTEM_NAME=Generic` + arm) and OFF +everywhere else, so desktop, the Max/C-ABI host builds (including Apple Silicon +arm64), and Hexagon are untouched. `double` always keeps Ooura regardless. The wrapper re-presents CMSIS in **Ooura's exact numeric contract** so nothing downstream changes and every intermediate spectrum matches the Ooura build to float epsilon: - **Sign convention.** CMSIS uses the engineering convention exp(−i2π/N); Ooura (and our documented packed layout) uses exp(+i2π/N). The wrapper - conjugates the imaginary bins on every transform. Verified by + conjugates the imaginary bins on every transform. Verified by DspTap's `real_fft_test/0.SignConventionIsPlusI` running on the CMSIS backend. - **Inverse scaling.** CMSIS's inverse RFFT is 1/N-normalized; Ooura's is unnormalized (the caller applies 2/N). The wrapper scales the CMSIS inverse @@ -79,7 +86,7 @@ float epsilon: trip. Both fold into passes the `_inplace` methods already do (~1% overhead). It is on automatically with the M55 toolchain; force the Ooura path with -`-DMUTAP_FFT_CMSIS=OFF`: +`-DTAP_DSP_FFT_CMSIS=OFF`: ```sh # CMSIS backend (default on M55) @@ -87,24 +94,40 @@ cmake -B build-m55 -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake \ -DCMAKE_BUILD_TYPE=Release # Ooura fallback cmake -B build-m55-ooura -DCMAKE_TOOLCHAIN_FILE=cmake/arm-cortex-m55-mps3.cmake \ - -DCMAKE_BUILD_TYPE=Release -DMUTAP_FFT_CMSIS=OFF + -DCMAKE_BUILD_TYPE=Release -DTAP_DSP_FFT_CMSIS=OFF ``` Forcing the option ON on a non-Arm processor is a hard error (Helium/NEON only). ### What is validated -- **`tests/test_fft_backend.cpp`** — asserts the CMSIS forward output matches a - direct Ooura `rdft_f` reference bin-for-bin (<5e-6 relative) and that the - round trip reproduces the input, at both certified sizes (512, 2048). Runs on - the M55-CMSIS leg (in `tests/bare_metal_main.cpp`'s selection). -- **The whole `test_fft.cpp` contract suite** (packing, +i sign convention, - Parseval, float-tracks-double) exercises `basic_real_fft`, so it - re-validates the CMSIS backend automatically when the option is on. -- **The full emulated float32 ITU battery** runs on the M55 with the CMSIS - backend (now the default): 58/58 tests pass — the AEC still meets every - asserted float32 gate on CMSIS FFTs. A dedicated CI leg re-runs the same - battery with `-DMUTAP_FFT_CMSIS=OFF` to keep the Ooura fallback honest. +- **DspTap's `tests/test_fft_backend.cpp`** (`CertifiedGeometries/ + fft_backend_parity`) asserts the CMSIS forward output matches a direct Ooura + `rdft_f` reference bin-for-bin (<5e-6 relative) and that the round trip + reproduces the input, at both certified sizes (512, 2048). Honest status: at + the current DspTap pin this gate on the CMSIS backend is executed by **no + CI** — the suites moved to DspTap with the FFT (so they are gone from + MuTap's `tests/bare_metal_main.cpp` selection), and DspTap's own M55 leg is + compile-only with its tests off; it will run once DspTap's embedded legs land + (DspTap #17, plan Part 10). Until then the float32 battery below is the only + CMSIS gate anywhere. +- **The whole DspTap `test_fft.cpp` contract suite** (packing, +i sign + convention, Parseval, float-tracks-double) exercises `basic_real_fft`, + so it re-validates the CMSIS backend automatically wherever it runs with the + option on (same status as above for the M55). +- **The emulated float32 battery** (`mutap_tests_emulated`, the 52-test + selection in `tests/bare_metal_main.cpp`) runs on the M55 with the CMSIS + backend (the default): 52/52 pass — the AEC still meets every asserted + float32 gate on CMSIS FFTs. A dedicated CI leg re-runs the same battery with + `-DTAP_DSP_FFT_CMSIS=OFF` to keep the Ooura fallback honest. +- **`tests/fingerprint_harness.cpp`** (`mutap_fingerprint`) prints one FNV-1a + fingerprint per (component, profile) over a fixed corpus on every CI leg, + including both M55 legs, so a pin-to-pin difference in any output sample is + visible as a diff of two logs; it is the bit-identity gate every DspTap pin + bump runs (procedure at the top of the file). Between the two M55 legs it + shows what the contract predicts — the seven `double` lines identical, the + seven `float` lines all different — which documents the backends' difference + but asserts nothing about CMSIS accuracy; that is the parity gate's job. ### Hexagon: deferred @@ -116,19 +139,26 @@ shape. Hexagon stays on scalar Ooura until an HVX FFT is available. ### Refreshing the vendored CMSIS subset -`third_party/cmsis-dsp/` is a minimal subset (8 sources + header closure), -pinned by commit in `third_party/cmsis-dsp/VENDOR.md`. To bump it: re-run the -`gcc -M` closure over the eight sources for `-mcpu=cortex-m55`, copy exactly the -files it opens, update `VENDOR.md`, then re-run `tests/test_fft_backend.cpp` and -the full float32 battery on the M55 leg. Do not hand-edit vendored sources. +DspTap's `third_party/cmsis-dsp/` is a minimal subset (8 sources + header +closure), pinned by commit in its `VENDOR.md`. To bump it (in DspTap): re-run +the `gcc -M` closure over the eight sources for `-mcpu=cortex-m55`, copy exactly +the files it opens, update `VENDOR.md`, then re-run DspTap's +`tests/test_fft_backend.cpp` and, after bumping the pin here, the full float32 +battery on the M55 leg. Do not hand-edit vendored sources. ## FFT backend: Apple vDSP on macOS -The same seam carries a second backend: on macOS the float32 real FFT routes -through Apple's **vDSP** (Accelerate) instead of Ooura (`MUTAP_FFT_ACCELERATE`, -default ON for Apple). This is the FFT the `mutap.aec~` Max external runs, and -it is the *only* fast FFT Apple Silicon gets — the CMSIS backend is scoped to -the bare-metal M55, so arm64 macOS was on Ooura before this. +The same seam carries a second backend: on macOS the float32 real FFT can +route through Apple's **vDSP** (Accelerate) instead of Ooura +(`TAP_DSP_FFT_ACCELERATE`, DspTap's default ON for Apple). It is the *only* +fast FFT Apple Silicon gets — the CMSIS backend is scoped to the bare-metal +M55, so arm64 macOS was on Ooura before this. **MuTap now turns it back OFF** +(root `CMakeLists.txt`, tap/MuTap#31): on spectra with exactly-empty bins the +alignment-selected vDSP kernel measured far less accurate than Ooura and the +G.168 tone rows failed on it, and Apple documents the routines as free to +rearrange arithmetic. The numbers below stand as the measurement behind that +decision; `-DTAP_DSP_FFT_ACCELERATE=ON` still selects it for anyone who wants +to re-measure. ### Measured (Apple Silicon, macOS CI runner) @@ -168,13 +198,15 @@ deployment precision *through vDSP itself*, not merely parity against Ooura. ### How it is wired / validated -Default ON for Apple (`APPLE` in CMake), OFF elsewhere, mutually exclusive with -the CMSIS backend; CMake links `-framework Accelerate` and defines the macro on -the `mutap` interface target. Force Ooura with `-DMUTAP_FFT_ACCELERATE=OFF`. -Validation is automatic: the `macos-latest` CI job builds with the backend on by -default, so `tests/test_fft_backend.cpp` (bin-for-bin vs Ooura `rdft_f`), the -whole `test_fft.cpp` contract suite, and the full float32 battery all run on -vDSP there. The vDSP setup's read-only twiddle tables are held by a shared_ptr +In DspTap: default ON for Apple (`APPLE` in CMake), OFF elsewhere, mutually +exclusive with the CMSIS backend; CMake links `-framework Accelerate` and +defines the macro on the `tap::dsp` interface target. MuTap's root +`CMakeLists.txt` sets `TAP_DSP_FFT_ACCELERATE` OFF (not FORCE), so an explicit +`-DTAP_DSP_FFT_ACCELERATE=ON` still wins. DspTap's `tests/test_fft_backend.cpp` +(bin-for-bin vs Ooura `rdft_f`) and `test_fft.cpp` contract suite validate the +backend in DspTap's own macOS CI; MuTap's `macos-latest` job gates the Ooura +float32 path it actually ships and records the backend it built. The vDSP +setup's read-only twiddle tables are held by a shared_ptr so `basic_real_fft` keeps value semantics; transforms are noexcept and allocation-free. @@ -201,6 +233,8 @@ Measured (icount, vs the branchy form on each target): So each target runs its faster form; neither regresses. Because it is per-target, no single build compiles both shapes — the `branchless-parity` CI -job compiles `tests/branchless_parity_check.cpp` once per macro value and diffs -an output fingerprint over a 600-block double-talk corpus, guaranteeing the two -forms stay sample-exact. The m55 baselines record the branch-free form. +job compiles the fingerprint harness (`tests/fingerprint_harness.cpp`, the +`mutap_fingerprint` target; see "What is validated" above) once per macro value +and diffs every `FINGERPRINT` line — 14 of them, seven components in both +profiles over the 400-block corpus — guaranteeing the two forms stay +sample-exact. The m55 baselines record the branch-free form. diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index af2d493..d3fd652 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -69,13 +69,31 @@ else() # of it the double-typed adaptive suites — target-independent math # already covered on every host platform. Run the same emulation-sized # selection as the Cortex-M55 leg (tests/bare_metal_main.cpp): the - # float32 typed suites, the double FFT, the LP/conditioning suites, the - # float closed-loop scenarios and the validation/contract tests. + # float32 typed suites, the double Kalman core, the LP/conditioning + # suites, the float closed-loop scenarios and the validation/contract + # tests. (The FFT suites left with the FFT: they run in DspTap's own CI.) set(mutap_emulated_selection "") if(CMAKE_CROSSCOMPILING) set(mutap_emulated_selection TEST_FILTER - "real_fft_test/0.*:real_fft_test/1.*:RealFftCrossPrecision.*:CertifiedGeometries/fft_backend_parity.*:fdaf_test/0.*:FdafCrossPrecision.*:FdafConfigValidation.*:FdafRtContract.*:fd_kalman_test/0.*:fd_kalman_test/1.*:kalman_loop_test/0.*:FdKalmanConfigValidation.*:FdKalmanRtContract.*:Levinson.*:LpcPredictor.*:SpeechPredictor.*:WarpedLpcPredictor.*:PredictorConfigValidation.*:pem_afc_test/0.*:PemAfcConfigValidation.*:PemAfcRtContract.*:closed_loop_test/0.*:burst_test/0.*:aec_test/0.*:AdaptationControlConfigValidation.*:nn_suppressor_test/0.*:NnSuppressorCrossPrecision.*:NnChainFloat32.*") + "fdaf_test/0.*:FdafCrossPrecision.*:FdafConfigValidation.*:FdafRtContract.*:fd_kalman_test/0.*:fd_kalman_test/1.*:kalman_loop_test/0.*:FdKalmanConfigValidation.*:FdKalmanRtContract.*:Levinson.*:LpcPredictor.*:SpeechPredictor.*:WarpedLpcPredictor.*:PredictorConfigValidation.*:pem_afc_test/0.*:PemAfcConfigValidation.*:PemAfcRtContract.*:closed_loop_test/0.*:burst_test/0.*:aec_test/0.*:AdaptationControlConfigValidation.*:nn_suppressor_test/0.*:NnSuppressorCrossPrecision.*:NnChainFloat32.*") endif() gtest_discover_tests(mutap_tests DISCOVERY_TIMEOUT 120 ${mutap_emulated_selection} PROPERTIES TIMEOUT 900) endif() + +# The bit-identity gate for every DspTap pin bump (tests/fingerprint_harness.cpp): +# one FNV-1a fingerprint per (component, profile) over a fixed synthetic corpus. +# A plain executable with its own main — no gtest — so the branchless-parity +# CI job can also compile the same file standalone. Registered as a test on +# every target so each CI log carries the lines; on bare metal the pass +# criterion is the completion line, as for mutap_tests_emulated. +add_executable(mutap_fingerprint fingerprint_harness.cpp) +target_link_libraries(mutap_fingerprint PRIVATE + MuTap::MuTap + mutap_warnings) +add_test(NAME mutap_fingerprint COMMAND mutap_fingerprint) +set_tests_properties(mutap_fingerprint PROPERTIES TIMEOUT 900) +if(MUTAP_BARE_METAL) + set_tests_properties(mutap_fingerprint PROPERTIES + PASS_REGULAR_EXPRESSION "FINGERPRINT_COMPLETE") +endif() diff --git a/tests/bare_metal_main.cpp b/tests/bare_metal_main.cpp index 40c5017..b7d05e1 100644 --- a/tests/bare_metal_main.cpp +++ b/tests/bare_metal_main.cpp @@ -2,7 +2,8 @@ // qemu-system-arm): there is no argv on the target, so the // emulation-appropriate selection is baked in. This is a POSITIVE filter: // the float32 typed suites (the embedded profile this target exists for), -// the double FFT (small; exercises the soft-float path), the LP / +// the double Kalman core (exercises the soft-float path; the FFT suites +// moved to DspTap with the FFT and run in its own CI), the LP / // conditioning suite, the float closed-loop scenarios including the PEM // canceller's tonal headline, the float-tracks-double oracle check, and // the learned suppressor's float profile with its own oracle check (the @@ -29,8 +30,6 @@ int main() { // Typed-suite naming: /0 = float, /1 = double (sample_types order). ::testing::GTEST_FLAG(filter) = - "real_fft_test/0.*:real_fft_test/1.*:RealFftCrossPrecision.*:" - "CertifiedGeometries/fft_backend_parity.*:" "fdaf_test/0.*:FdafCrossPrecision.*:FdafConfigValidation.*:FdafRtContract.*:" "fd_kalman_test/0.*:fd_kalman_test/1.*:FdKalmanConfigValidation.*:FdKalmanRtContract.*:" "Levinson.*:LpcPredictor.*:SpeechPredictor.*:WarpedLpcPredictor.*:PredictorConfigValidation.*:" @@ -44,8 +43,8 @@ int main() { // A filter typo selects zero tests and RUN_ALL_TESTS() returns 0 — an // empty run must not pass green. Checked after the run because gtest // only applies the filter inside RUN_ALL_TESTS. The on-target selection - // is ~57 tests; 30 leaves headroom for legitimate removals without - // masking a typo. + // is 52 tests (the FFT suites left with the FFT); 30 leaves headroom + // for legitimate removals without masking a typo. const int selected = ::testing::UnitTest::GetInstance()->test_to_run_count(); if (selected < 30) { std::printf("only %d tests selected (expected >= 30): filter is broken\n", selected); diff --git a/tests/branchless_parity_check.cpp b/tests/branchless_parity_check.cpp deleted file mode 100644 index a9e35c3..0000000 --- a/tests/branchless_parity_check.cpp +++ /dev/null @@ -1,65 +0,0 @@ -// SPDX-License-Identifier: MIT -// Copyright 2026 MuTap contributors -// -// Bit-identity guard for the suppressor's two pass-1 forms (see -// MUTAP_SUPPRESSOR_BRANCHLESS in include/mutap/postfilter.h). The branch-free -// form is compiled on Arm Helium and the branchy form everywhere else, so no -// single build exercises both; CI compiles THIS file twice — once per macro -// value — and diffs the fingerprint below. Identical fingerprints prove the -// two forms produce sample-exact output, which is what lets the branch-free -// form ride the compliance battery that certified the branchy one. -// -// Standalone (not part of mutap_tests): the whole point is to build it once -// per macro value. See the "branchless parity" step in .github/workflows/ci.yml. -#include -#include -#include -#include - -#include "mutap/postfilter.h" - -int main() { - auto chain = tap::mu::aec_chain(tap::mu::aec_chain_preset(256, 8, 48000.0)); - constexpr int B = 256; - std::vector x(B), y(B), o(B); - - std::uint32_t s = 0x12345u; - auto nx = [&s]() noexcept { - s ^= s << 13; - s ^= s >> 17; - s ^= s << 5; - return static_cast(s) / 2147483648.0f - 1.0f; - }; - - // FNV-1a over the raw bits of every output sample: any single-ULP - // divergence between the two forms flips the fingerprint. - std::uint64_t fp = 1469598103934665603ULL; - auto mix = [&fp](float v) noexcept { - std::uint32_t b; - __builtin_memcpy(&b, &v, 4); - fp = (fp ^ b) * 1099511628211ULL; - }; - - // A corpus varied enough to drive every data-dependent branch the two - // forms disagree on if they ever diverge: level sweeps, double-talk - // bursts (near-end incoherent with the echo), and quiet stretches. - for (int blk = 0; blk < 600; ++blk) { - const float amp = 0.05f + 0.1f * std::fabs(std::sin(blk * 0.03f)); - const bool dt = (blk / 40) % 3 == 0; - for (int i = 0; i < B; ++i) { - const float far = amp * nx(); - const float echo = 0.3f * far; - const float near = dt ? 0.08f * std::sin((blk * B + i) * 0.05f) + 0.03f * nx() : 0.0f; - x[i] = far; - y[i] = echo + near + 0.001f * nx(); - } - chain.process_block(x.data(), y.data(), o.data()); - for (float v : o) { - mix(v); - } - } - - std::printf("MUTAP_SUPPRESSOR_PARITY_FP %016llx branchless=%d\n", static_cast(fp), - MUTAP_SUPPRESSOR_BRANCHLESS); - return 0; -} diff --git a/tests/fingerprint_harness.cpp b/tests/fingerprint_harness.cpp new file mode 100644 index 0000000..3e05b88 --- /dev/null +++ b/tests/fingerprint_harness.cpp @@ -0,0 +1,343 @@ +// SPDX-License-Identifier: MIT +// Copyright 2026 MuTap contributors +// +// THE BIT-IDENTITY GATE FOR EVERY DspTap PIN BUMP. +// +// Runs a fixed, xorshift-driven synthetic echo corpus through each of MuTap's +// signal-processing components in BOTH numeric profiles and prints one line +// per (component, profile): +// +// FINGERPRINT <16 hex digits> +// +// where the digits are FNV-1a (64-bit) over the raw bytes of every output +// sample the component produced. Any single-ULP change anywhere in a +// component's output stream flips its line. Components: +// +// fdaf partitioned_fdaf, the NLMS core with its control stack +// fd_kalman partitioned_fdkf at the certified preset calibration +// pem_afc the FDAF-PEM-AFROW canceller (speech predictor + fdaf) +// postfilter residual_suppressor alone, fed a fixed (e, yhat) pair +// nn_suppressor the learned post-filter alone, deterministic weights +// aec_chain the certified chain: fd_kalman + postfilter (preset) +// aec_chain_nn the learned chain: fd_kalman + nn_suppressor (preset) +// +// For fdaf and fd_kalman both output channels are hashed: the error block +// and the echo-estimate block (the four-argument process_block), because +// error = desired - estimate can absorb a one-ULP move of the estimate when +// the near end is loud; the estimate stream alone exposes it. +// +// The corpus is generated in double from integer state with basic IEEE +// arithmetic only (no libm, no , no wall clock, no filesystem) and +// rounded once to float, so both profiles consume identical sample values +// and the harness runs unchanged on bare metal. Determinism holds per +// (host, compiler, flags): libm and fp-contraction differ across hosts, so +// two fingerprints are comparable only when both runs were produced by the +// same build configuration on the same machine — which is exactly the pin- +// bump workflow below. The '#' header names the float32 FFT backend the +// binary was compiled with (backend=cmsis|vdsp|ooura), so a log is +// self-describing and the M55 legs can assert which backend they ran. +// +// How to diff two DspTap pins (the check every submodule bump runs): +// +// cmake -S . -B build -DCMAKE_BUILD_TYPE=Release +// cmake --build build --target mutap_fingerprint +// ./build/tests/mutap_fingerprint > /tmp/before.txt +// git -C submodules/dsptap checkout +// cmake --build build --target mutap_fingerprint +// ./build/tests/mutap_fingerprint > /tmp/after.txt +// diff /tmp/before.txt /tmp/after.txt +// +// An empty diff is the proof that the bump changed no output sample; every +// stage of the FFT plan (DspTap docs/audit-fft-and-code-smells.md) states +// which lines here must be unchanged and which are expected to move (the +// float profile through an FFT port; never the double golden model unless +// the plan says so). CI runs this binary on every leg, hosted and emulated +// (a "Fingerprints" step records the lines in each log, so two CI logs can +// be diffed the same way, and the M55's CMSIS and Ooura legs against each +// other) and once more, compiled twice, as the suppressor's +// branch-free/branchy parity check (MUTAP_SUPPRESSOR_BRANCHLESS, see +// include/mutap/postfilter.h): the two builds must print identical lines. +// +// Builds two ways: as the normal CMake target mutap_fingerprint (a ctest +// test on every target), and standalone as the parity job compiles it (g++ +// on this file plus DspTap's Ooura .c files — see the branchless-parity job +// in .github/workflows/ci.yml). +#include +#include +#include +#include +#include +#include +#include + +#include "mutap/fd_kalman.h" +#include "mutap/fdaf.h" +#include "mutap/nn_chain.h" +#include "mutap/pem_afc.h" +#include "mutap/postfilter.h" + +namespace { + + // The certified reference geometry: block 256, 8 partitions (2048 taps) + // at 48 kHz — the calibration point of every preset. + constexpr std::size_t k_block = 256; + constexpr std::size_t k_partitions = 8; + constexpr double k_sample_rate = 48000.0; + // Corpus length in blocks (~2.1 s of audio). Long enough to carry every + // canceller through convergence and into the double-talk / quiet + // segments; short enough for the double profile under soft-float + // emulation on the Cortex-M legs. + constexpr std::size_t k_blocks = 400; + + // The float32 FFT backend this binary was compiled against, from the + // compile definitions DspTap's fft.h keys on (not the CMake cache). +#if defined(TAP_DSP_FFT_CMSIS) + constexpr const char* k_backend = "cmsis"; +#elif defined(TAP_DSP_FFT_ACCELERATE) + constexpr const char* k_backend = "vdsp"; +#else + constexpr const char* k_backend = "ooura"; +#endif + + /// xorshift32 -> uniform in [-1, 1), exactly representable in float: + /// a 24-bit signed integer times a power of two, so the corpus is the + /// same value set in both profiles by construction. + class xorshift32 { + public: + explicit xorshift32(std::uint32_t seed) noexcept + : m_s(seed) {} + + double next() noexcept { + m_s ^= m_s << 13; + m_s ^= m_s >> 17; + m_s ^= m_s << 5; + const auto q = static_cast(m_s >> 8) - 8388608; // [-2^23, 2^23) + return static_cast(q) * (1.0 / 8388608.0); + } + + private: + std::uint32_t m_s; + }; + + /// FNV-1a over raw sample bytes. Any ULP flips it. + class fingerprint { + public: + template + void mix(const Sample* v, std::size_t n) noexcept { + unsigned char bytes[sizeof(Sample)]; + for (std::size_t i = 0; i < n; ++i) { + std::memcpy(bytes, &v[i], sizeof(Sample)); + for (const unsigned char b : bytes) { + m_h = (m_h ^ b) * 1099511628211ULL; + } + } + } + + std::uint64_t value() const noexcept { return m_h; } + + private: + std::uint64_t m_h = 1469598103934665603ULL; + }; + + /// One block of the corpus, produced on demand from fixed integer state + /// (no stored corpus: the bare-metal targets are RAM-bound). Every + /// component gets a fresh source, so every component sees the same + /// blocks in the same order. + /// + /// x far end: uniform noise under a slow triangular level sweep + /// y mic: 3-tap sparse echo of x + near end + a tiny floor + /// e a converged canceller's residual (y minus 95 % of the echo) + /// yhat that canceller's echo estimate (the echo itself) + /// + /// The near end is present in every third 40-block stretch from block + /// 40 on (double talk), so each canceller first converges in single + /// talk: one-pole-colored noise plus a white component, incoherent with + /// the echo — enough structure to drive every data-dependent path the + /// suppressors and the control stacks branch on. + template + class corpus_source { + public: + corpus_source() + : m_far(0x12345u) + , m_near(0x9E3779B9u) + , m_floor(0x2545F491u) + , m_history(k_history, 0.0) + , m_x(k_block) + , m_y(k_block) + , m_e(k_block) + , m_yhat(k_block) {} + + void generate(std::size_t blk) noexcept { + const std::size_t p = blk % 200; + const double tri = static_cast(p < 100 ? p : 200 - p) * 0.01; + const double amp = 0.05 + 0.1 * tri; + const bool dt = (blk / 40) % 3 == 1; + for (std::size_t i = 0; i < k_block; ++i) { + const double far = amp * m_far.next(); + // Delay line: newest sample at index 0. + for (std::size_t k = k_history - 1; k > 0; --k) { + m_history[k] = m_history[k - 1]; + } + m_history[0] = far; + const double echo = + 0.25 * m_history[k_block / 2] - 0.12 * m_history[k_block] + 0.06 * m_history[3 * k_block / 2]; + double near = 0.0; + if (dt) { + m_colored = 0.9 * m_colored + 0.1 * m_near.next(); + near = 0.6 * m_colored + 0.03 * m_near.next(); + } + const double floor = 0.001 * m_floor.next(); + // Round once to float; the double profile widens exactly. + m_x[i] = static_cast(static_cast(far)); + m_y[i] = static_cast(static_cast(echo + near + floor)); + m_e[i] = static_cast(static_cast(0.05 * echo + near + floor)); + m_yhat[i] = static_cast(static_cast(echo)); + } + } + + const Sample* x() const noexcept { return m_x.data(); } + const Sample* y() const noexcept { return m_y.data(); } + const Sample* e() const noexcept { return m_e.data(); } + const Sample* yhat() const noexcept { return m_yhat.data(); } + + private: + static constexpr std::size_t k_history = 3 * k_block / 2 + 1; + xorshift32 m_far; + xorshift32 m_near; + xorshift32 m_floor; + std::vector m_history; + double m_colored = 0.0; + std::vector m_x; + std::vector m_y; + std::vector m_e; + std::vector m_yhat; + }; + + /// Deterministic live weights at the shipping 48 kHz geometry (the + /// values do not matter, only that the network's gains vary with the + /// input and never change between runs). + tap::mu::nn_suppressor_weights nn_weights() { + const tap::mu::nn_geometry g{48000.0, 256, 26, 64, 96}; + xorshift32 rng(0xC0FFEE11u); + auto fill = [&rng](std::vector& v, std::size_t n) { + v.resize(n); + for (auto& w : v) { + w = static_cast(rng.next() * 0.3); + } + }; + tap::mu::nn_suppressor_weights w; + w.geometry = g; + fill(w.dense_in_w, g.dense * g.features()); + fill(w.dense_in_b, g.dense); + fill(w.gru_w_ih, 3 * g.gru * g.dense); + fill(w.gru_w_hh, 3 * g.gru * g.gru); + fill(w.gru_b_ih, 3 * g.gru); + fill(w.gru_b_hh, 3 * g.gru); + fill(w.dense_out_w, g.bands * g.gru); + fill(w.dense_out_b, g.bands); + return w; + } + + template + const char* profile_name() noexcept { + if constexpr (std::is_same_v) { + return "float"; + } + else { + static_assert(std::is_same_v, "two profiles: float and double"); + return "double"; + } + } + + /// Runs `step(source, out, estimate)` once per block and prints the + /// component's line. `out` is always hashed; `estimate` is hashed too + /// when the step writes it (the two cancellers' second output channel). + template + void run(const char* component, bool hashes_estimate, Step&& step) { + corpus_source src; + std::vector out(k_block); + std::vector estimate(k_block); + fingerprint fp; + for (std::size_t blk = 0; blk < k_blocks; ++blk) { + src.generate(blk); + step(src, out.data(), estimate.data()); + fp.mix(out.data(), k_block); + if (hashes_estimate) { + fp.mix(estimate.data(), k_block); + } + } + std::printf("FINGERPRINT %s %s %016llx\n", component, profile_name(), + static_cast(fp.value())); + } + + template + void run_profile() { + using namespace tap::mu; + using src_t = corpus_source; + + { + typename partitioned_fdaf::config cfg; + cfg.block_size = k_block; + cfg.partitions = k_partitions; + cfg.ipc_step_scaling = true; + cfg.ipc_freeze_threshold = Sample(0.1); + cfg.transient_freeze_ratio = Sample(8); + partitioned_fdaf f(cfg); + run("fdaf", true, + [&f](const src_t& s, Sample* out, Sample* est) { f.process_block(s.x(), s.y(), out, est); }); + } + { + partitioned_fdkf f(aec_chain_preset(k_block, k_partitions, k_sample_rate).canceller); + run("fd_kalman", true, + [&f](const src_t& s, Sample* out, Sample* est) { f.process_block(s.x(), s.y(), out, est); }); + } + { + typename pem_afc::config cfg; + cfg.fdaf.block_size = k_block; + cfg.fdaf.partitions = k_partitions; + pem_afc f(cfg); + run("pem_afc", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.x(), s.y(), out); }); + } + { + auto cfg = aec_chain_preset(k_block, k_partitions, k_sample_rate).postfilter; + cfg.block_size = k_block; + residual_suppressor f(cfg); + run("postfilter", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.e(), s.yhat(), out); }); + } + { + auto cfg = aec_chain_nn_preset(k_block, k_partitions, k_sample_rate, nn_weights()).postfilter; + nn_suppressor f(std::move(cfg)); + run("nn_suppressor", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.e(), s.yhat(), out); }); + } + { + aec_chain f(aec_chain_preset(k_block, k_partitions, k_sample_rate)); + run("aec_chain", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.x(), s.y(), out); }); + } + { + aec_chain_nn f(aec_chain_nn_preset(k_block, k_partitions, k_sample_rate, nn_weights())); + run("aec_chain_nn", false, + [&f](const src_t& s, Sample* out, Sample*) { f.process_block(s.x(), s.y(), out); }); + } + } + +} // namespace + +int main() { + // Informational header (not part of the diffed lines): the geometry, the + // float32 FFT backend and the suppressor form this binary compiled, so a + // log is self-describing and the M55 legs can assert their backend. + std::printf("# mutap_fingerprint block=%u partitions=%u rate=%u blocks=%u backend=%s branchless=%d\n", + static_cast(k_block), static_cast(k_partitions), + static_cast(k_sample_rate), static_cast(k_blocks), k_backend, + MUTAP_SUPPRESSOR_BRANCHLESS); + run_profile(); + run_profile(); + // CTest's pass criterion on bare metal, where semihosting does not + // reliably propagate the exit code: printed only after every line. + std::printf("FINGERPRINT_COMPLETE\n"); + return 0; +}