Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -502,7 +502,8 @@ jobs:
-DCMAKE_BUILD_TYPE=Release
-DSRT_BUILD_BENCHMARKS=ON
-DSRT_BUILD_COMPARE_BENCH=ON
&& cmake --build build-host -j 4 --target srt_bench_compare
-DSRT_BUILD_COMPARE_SHIM=ON
&& cmake --build build-host -j 4 --target srt_bench_compare srt_r8b_shim

- name: Build M55 comparison workload
run: >
Expand All @@ -512,6 +513,7 @@ jobs:
-DSRT_BUILD_TESTS=OFF -DSRT_BUILD_EXAMPLES=OFF
-DSRT_BUILD_ICOUNT_BENCH=ON -DSRT_ICOUNT_COMPARE=ON
&& cmake --build build-m55 -j 4 --target cmp_icount_lsr_medium
cmp_icount_srt_q15 cmp_icount_r8b_120

clang-format:
name: clang-format
Expand Down
8 changes: 6 additions & 2 deletions .github/workflows/compare.yml
Original file line number Diff line number Diff line change
Expand Up @@ -58,7 +58,10 @@ jobs:
-DSRT_BUILD_TESTS=OFF -DSRT_BUILD_EXAMPLES=OFF \
-DSRT_BUILD_ICOUNT_BENCH=ON -DSRT_ICOUNT_COMPARE=ON
cmake --build build-$tgt -j 4
for bin in cmp_icount_srt_float cmp_icount_lsr_medium cmp_icount_lsr_best; do
# Each engine at 2 s and 4 s: the difference is steady state, the
# remainder construction (bench/icount/cmp_main.cpp).
for bin in $(for e in srt_float srt_q15 lsr_medium lsr_best r8b_120 r8b_120_tb8; do
echo cmp_icount_$e cmp_icount_${e}_4s; done); do
out=$(qemu-system-arm -M $machine -nographic -semihosting \
-d plugin -plugin /tmp/libinsncount.so \
-kernel build-$tgt/bench/icount/$bin 2>&1)
Expand Down Expand Up @@ -122,7 +125,8 @@ jobs:
-DSRT_BUILD_TESTS=OFF -DSRT_BUILD_EXAMPLES=OFF \
-DSRT_BUILD_ICOUNT_BENCH=ON -DSRT_ICOUNT_COMPARE=ON
cmake --build build-hex -j 4
for bin in cmp_icount_srt_float cmp_icount_lsr_medium cmp_icount_lsr_best; do
for bin in $(for e in srt_float srt_q15 lsr_medium lsr_best r8b_120 r8b_120_tb8; do
echo cmp_icount_$e cmp_icount_${e}_4s; done); do
out=$(qemu-hexagon -d plugin -plugin /tmp/libinsncount.so \
build-hex/bench/icount/$bin 2>&1)
echo "$out" | grep -q 'SRT_ICOUNT_DONE ok=1' || {
Expand Down
8 changes: 8 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -77,3 +77,11 @@ option(SRT_BUILD_CAPI "Build the C ABI shared library" OFF)
if(SRT_BUILD_CAPI)
add_subdirectory(tools/capi)
endif()

# ctypes shim over r8brain-free-src for the comparison notebook (fetched at a
# commit pin; comparison-only, never linked into the library). See
# notebooks/asrc_comparison.ipynb and docs/COMPARISON.md.
option(SRT_BUILD_COMPARE_SHIM "Build the r8brain shim for the comparison notebook" OFF)
if(SRT_BUILD_COMPARE_SHIM)
add_subdirectory(tools/compare_shim)
endif()
14 changes: 9 additions & 5 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -218,8 +218,8 @@ sample-granular transfer, 0.5 FS sine, 1 s analysis window after settling):
| `transparent()` (L=512, T=80) | 133 dB | — | — | 108 dB | 0.83 ms |

AES17-style THD+N measured under identical conditions against
libsamplerate, soxr and hardware datasheet figures:
[docs/COMPARISON.md](docs/COMPARISON.md) (−132 dB THD+N / 149 dB DR at the
libsamplerate, soxr, r8brain-free-src and hardware datasheet figures:
[docs/COMPARISON.md](docs/COMPARISON.md) (−134 dB THD+N / 149 dB DR at the
24-bit interface, servo in the loop;
[notebooks/asrc_comparison.ipynb](notebooks/asrc_comparison.ipynb)).

Expand Down Expand Up @@ -311,9 +311,10 @@ two USB audio dongles, a Pi + Pico 2, two Pis over Ethernet), see
Methodology, optimization roadmap and regression gating live in
[docs/PERFORMANCE.md](docs/PERFORMANCE.md). Build the benchmarks with
`-DSRT_BUILD_BENCHMARKS=ON` (host only). A measured computational
head-to-head against libsamplerate and soxr — host wall-clock and embedded
instruction counts (`-DSRT_BUILD_COMPARE_BENCH=ON`, `SRT_ICOUNT_COMPARE`) —
lives in [docs/COMPARISON.md](docs/COMPARISON.md).
head-to-head against libsamplerate, soxr and r8brain-free-src — host
wall-clock and embedded instruction counts, steady state and construction
(`-DSRT_BUILD_COMPARE_BENCH=ON`, `SRT_ICOUNT_COMPARE`) — lives in
[docs/COMPARISON.md](docs/COMPARISON.md).

<!-- ICOUNT:BEGIN -->
Executed instructions per fixed workload (`bench/icount/`), measured under QEMU with a counting plugin — deterministic, and gated in CI at ±3% against `bench/baselines.json`:
Expand Down Expand Up @@ -440,3 +441,6 @@ window design (Kaiser 1974), band-limited interpolation (J. O. Smith,
CCRMA), polyphase decomposition and the harris length estimate, and textbook
2nd-order PLL servo design. No third-party source was copied. GoogleTest
(BSD-3) is fetched for tests only and is not part of the shipped headers.
r8brain-free-src (MIT) is fetched at a commit pin only when the opt-in
comparison builds are enabled (`cmake/r8brain.cmake`); it is never linked
into the library or its tests.
11 changes: 7 additions & 4 deletions bench/compare/CMakeLists.txt
Original file line number Diff line number Diff line change
@@ -1,15 +1,18 @@
# Head-to-head computational comparison against general-purpose resamplers
# (libsamplerate, soxr) at a fixed, known near-unity ratio. Host-only:
# the competitor libraries come from the system (pkg-config). See
# docs/COMPARISON.md for the methodology and the measured numbers.
# (libsamplerate, soxr, r8brain-free-src) at a fixed, known near-unity ratio.
# Host-only: libsamplerate and soxr come from the system (pkg-config);
# r8brain is header-only and fetched at a commit pin (cmake/r8brain.cmake).
# See docs/COMPARISON.md for the methodology and the measured numbers.
find_package(PkgConfig REQUIRED)
pkg_check_modules(SAMPLERATE REQUIRED IMPORTED_TARGET samplerate)
pkg_check_modules(SOXR REQUIRED IMPORTED_TARGET soxr)
include(${PROJECT_SOURCE_DIR}/cmake/r8brain.cmake)

add_executable(srt_bench_compare bench_compare.cpp)
target_link_libraries(srt_bench_compare PRIVATE
SampleRateTap::SampleRateTap
srt_warnings
benchmark::benchmark_main
PkgConfig::SAMPLERATE
PkgConfig::SOXR)
PkgConfig::SOXR
srt_r8brain)
75 changes: 73 additions & 2 deletions bench/compare/bench_compare.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -11,12 +11,22 @@
//
// Quality pairing (stopband attenuation, vendor-stated):
// srt balanced (120 dB) ~ libsamplerate MEDIUM (121 dB) ~ soxr HQ (~120 dB)
// ~ r8brain ReqAtten=120 (default 2% transition band,
// and 8%: the lowest-latency setting flat to 20 kHz)
// srt transparent (140 dB) ~ libsamplerate BEST (144 dB) ~ soxr VHQ (~170 dB)
// ~ r8brain CDSPResampler16 (136.45 dB)
// plus r8brain's CDSPResampler24 (180.15 dB), its preset for 24-bit/float work.
//
// r8brain is mono per instance with double-precision I/O, so the harness runs
// one instance per channel and pays the float<->double (de)interleave inside
// the timed loop — the cost any float-interleaved caller pays to use it.
#include <cmath>
#include <cstddef>
#include <memory>
#include <numbers>
#include <vector>

#include <CDSPResampler.h>
#include <benchmark/benchmark.h>
#include <samplerate.h>
#include <soxr.h>
Expand Down Expand Up @@ -174,6 +184,37 @@ namespace {
state.SetItemsProcessed(frames);
}

void r8bBench(benchmark::State& state, double reqAtten, double transBandPct, std::size_t channels) {
std::vector<std::unique_ptr<r8b::CDSPResampler>> rs;
for (std::size_t c = 0; c < channels; ++c)
rs.push_back(std::make_unique<r8b::CDSPResampler>(48000.0, 48000.0 * kRatio, static_cast<int>(kBlock),
transBandPct, reqAtten, r8b::fprLinearPhase));
InputTap in(48000, channels);
std::vector<float> inBlock(kBlock * channels);
std::vector<double> chIn(kBlock);
std::vector<float> out(4 * kBlock * channels);

std::int64_t frames = 0;
for (auto _ : state) {
in.pop(inBlock.data(), kBlock);
int got = 0;
for (std::size_t c = 0; c < channels; ++c) {
for (std::size_t i = 0; i < kBlock; ++i)
chIn[i] = inBlock[i * channels + c];
double* op = nullptr;
got = rs[c]->process(chIn.data(), static_cast<int>(kBlock), op);
for (int i = 0; i < got; ++i)
out[static_cast<std::size_t>(i) * channels + c] = static_cast<float>(op[i]);
}
benchmark::DoNotOptimize(out.data());
frames += got;
}
// Input frames consumed before the first output frame appears: r8brain
// hides its filter delay by withholding output, so this is its latency.
state.counters["latency_frames"] = rs[0]->getInLenBeforeOutPos(0);
state.SetItemsProcessed(frames);
}

// --- ~120 dB tier: mono / stereo / 8ch -------------------------------------
void BM_SRT_Balanced_1ch(benchmark::State& s) {
srtBench<float>(s, tap::samplerate::filter_spec::balanced(), 1);
Expand Down Expand Up @@ -211,6 +252,28 @@ namespace {
BENCHMARK(BM_SOXR_HQ_1ch);
BENCHMARK(BM_SOXR_HQ_2ch);
BENCHMARK(BM_SOXR_HQ_8ch);
void BM_R8B_120dB_1ch(benchmark::State& s) {
r8bBench(s, 120.0, 2.0, 1);
}
void BM_R8B_120dB_2ch(benchmark::State& s) {
r8bBench(s, 120.0, 2.0, 2);
}
void BM_R8B_120dB_8ch(benchmark::State& s) {
r8bBench(s, 120.0, 2.0, 8);
}
BENCHMARK(BM_R8B_120dB_1ch);
BENCHMARK(BM_R8B_120dB_2ch);
BENCHMARK(BM_R8B_120dB_8ch);
// r8brain's default 2% transition band keeps its passband flat far past
// 20 kHz and pays for it in delay. The passband-matched row takes the
// lowest-latency linear-phase setting still flat to 20 kHz like srt
// balanced, from the sweep in notebooks/asrc_comparison.ipynb: 8% (200
// input frames; latency is not monotonic in the knob — 10% costs 212 —
// and 12% already droops 0.11 dB at 20 kHz).
void BM_R8B_120dB_TB8_2ch(benchmark::State& s) {
r8bBench(s, 120.0, 8.0, 2);
}
BENCHMARK(BM_R8B_120dB_TB8_2ch);

// --- ~140 dB tier, stereo ---------------------------------------------------
void BM_SRT_Transparent_2ch(benchmark::State& s) {
Expand All @@ -225,9 +288,17 @@ namespace {
BENCHMARK(BM_SRT_Transparent_2ch);
BENCHMARK(BM_LSR_Best_2ch);
BENCHMARK(BM_SOXR_VHQ_2ch);
void BM_R8B_16bit_2ch(benchmark::State& s) {
r8bBench(s, 136.45, 2.0, 2); // CDSPResampler16's preset attenuation
}
void BM_R8B_24bit_2ch(benchmark::State& s) {
r8bBench(s, 180.15, 2.0, 2); // CDSPResampler24's preset attenuation
}
BENCHMARK(BM_R8B_16bit_2ch);
BENCHMARK(BM_R8B_24bit_2ch);

// --- Fixed-point (no competitor analog; libsamplerate and soxr are
// float-only engines — this is the row embedded targets actually run) ------
// --- Fixed-point (no competitor analog; libsamplerate, soxr and r8brain
// are floating-point engines — this is the row embedded targets actually run) ------
void BM_SRT_Q15_Balanced_2ch(benchmark::State& s) {
srtBench<std::int16_t>(s, tap::samplerate::filter_spec::balanced(), 2);
}
Expand Down
49 changes: 32 additions & 17 deletions bench/icount/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,7 @@ foreach(_sc IN LISTS _srt_icount_scenarios)
endforeach()

# Cross-resampler comparison workloads (docs/COMPARISON.md): same fixed
# workload through our datapath and through libsamplerate, built for the
# workload through our datapath and through libsamplerate and r8brain, built for the
# same target. Named cmp_icount_* so the ratchet's srt_icount_* glob never
# picks them up — competitor counts are recorded in docs, not gated.
option(SRT_ICOUNT_COMPARE "Build resampler-comparison icount workloads" OFF)
Expand All @@ -39,21 +39,36 @@ if(SRT_ICOUNT_COMPARE)
URL_HASH SHA256=3258da280511d24b49d6b08615bbe824d0cacc9842b0e4caf11c52cf2b043893)
FetchContent_MakeAvailable(libsamplerate)

add_executable(cmp_icount_srt_float cmp_main.cpp)
target_compile_definitions(cmp_icount_srt_float PRIVATE SRT_CMP_ENGINE=0)
target_link_libraries(cmp_icount_srt_float PRIVATE
SampleRateTap::SampleRateTap srt_warnings)

foreach(_eng IN ITEMS 1 2)
if(_eng EQUAL 1)
set(_name lsr_medium)
else()
set(_name lsr_best)
endif()
add_executable(cmp_icount_${_name} cmp_main.cpp)
target_compile_features(cmp_icount_${_name} PRIVATE cxx_std_20)
target_compile_definitions(cmp_icount_${_name} PRIVATE SRT_CMP_ENGINE=${_eng})
# No srt_warnings: samplerate.h is third-party.
target_link_libraries(cmp_icount_${_name} PRIVATE samplerate)
# name:engine. Each engine is built twice, at the default 2 s and at 4 s
# (suffix _4s): the difference is the steady-state cost, the remainder is
# one-time construction (cmp_main.cpp's header).
set(_srt_cmp_engines srt_float:0 lsr_medium:1 lsr_best:2 r8b_120:3 r8b_120_tb8:4 srt_q15:5)
# r8brain-free-src, same pin as bench/compare. It compiles bare-metal only
# with a no-op std::mutex supplied for thread-less newlib toolchains
# (r8b_single_thread_mutex.h says why that is exact for this workload).
include(${PROJECT_SOURCE_DIR}/cmake/r8brain.cmake)
foreach(_e IN LISTS _srt_cmp_engines)
string(REPLACE ":" ";" _parts "${_e}")
list(GET _parts 0 _name)
list(GET _parts 1 _eng)
foreach(_secs IN ITEMS 2 4)
set(_bin cmp_icount_${_name})
if(_secs EQUAL 4)
set(_bin ${_bin}_4s)
endif()
add_executable(${_bin} cmp_main.cpp)
target_compile_features(${_bin} PRIVATE cxx_std_20)
target_compile_definitions(${_bin} PRIVATE SRT_CMP_ENGINE=${_eng} SRT_CMP_SECONDS=${_secs})
if(_eng EQUAL 0 OR _eng EQUAL 5)
target_link_libraries(${_bin} PRIVATE SampleRateTap::SampleRateTap srt_warnings)
elseif(_eng LESS_EQUAL 2)
# No srt_warnings: samplerate.h is third-party.
target_link_libraries(${_bin} PRIVATE samplerate)
else()
target_compile_options(${_bin} PRIVATE -include ${CMAKE_CURRENT_SOURCE_DIR}/r8b_single_thread_mutex.h)
# No srt_warnings: the r8brain headers are third-party.
target_link_libraries(${_bin} PRIVATE srt_r8brain)
endif()
endforeach()
endforeach()
endif()
Loading
Loading