From b60bb585fdc69fd7f7c8e881028137f575972164 Mon Sep 17 00:00:00 2001 From: Nikolaus Heger Date: Thu, 10 Sep 2026 14:08:00 +0800 Subject: [PATCH] Make mining benchmarks measure full nonce ranges Use the maximum U512 difficulty for throughput measurements so solutions are effectively unreachable and each search processes its complete range. Prepare job contexts outside timed Criterion loops, use exact inclusive range sizes, and remove the early-exit solution-finding benchmark. --- crates/engine-cpu/benches/cpu_engine_bench.rs | 24 ++- crates/engine-gpu/benches/gpu_engine_bench.rs | 140 +++--------------- crates/engine-gpu/examples/hashrate.rs | 2 +- crates/miner-cli/src/main.rs | 6 +- 4 files changed, 35 insertions(+), 137 deletions(-) diff --git a/crates/engine-cpu/benches/cpu_engine_bench.rs b/crates/engine-cpu/benches/cpu_engine_bench.rs index 323776e0..7c8a0920 100644 --- a/crates/engine-cpu/benches/cpu_engine_bench.rs +++ b/crates/engine-cpu/benches/cpu_engine_bench.rs @@ -5,24 +5,27 @@ use primitive_types::U512; use rand::RngCore; use std::sync::atomic::AtomicBool; +const BENCHMARK_DIFFICULTY: U512 = U512::MAX; + +fn benchmark_context() -> JobContext { + let mut header = [0u8; 32]; + rand::rng().fill_bytes(&mut header); + JobContext::new(header, BENCHMARK_DIFFICULTY) +} + fn bench_cpu_fast_engine(c: &mut Criterion) { - // Create the engine with batch size of 10000 let engine = FastCpuEngine::new(10_000); let cancel_flag = AtomicBool::new(false); let cancel_check = AtomicBoolCancelCheck(&cancel_flag); + let ctx = benchmark_context(); let large_range = Range { start: U512::from(0u64), - end: U512::from(100000u64), // 100,000 nonces + end: U512::from(99_999u64), }; c.bench_function("cpu_fast_large_range", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(10_000_000u64); - let ctx = JobContext::new(header, difficulty); - let result = engine.search_range( black_box(&ctx), black_box(large_range.clone()), @@ -34,13 +37,8 @@ fn bench_cpu_fast_engine(c: &mut Criterion) { } fn bench_hash_from_nonce(c: &mut Criterion) { - // Create a test job context - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(1000u64); - let ctx = JobContext::new(header, difficulty); + let ctx = benchmark_context(); - // Create some test nonce values let test_nonce_values: Vec = (0..100).map(|i| U512::from(1000u64 + i)).collect(); c.bench_function("hash_from_nonce_single", |b| { diff --git a/crates/engine-gpu/benches/gpu_engine_bench.rs b/crates/engine-gpu/benches/gpu_engine_bench.rs index 97344b87..69a52feb 100644 --- a/crates/engine-gpu/benches/gpu_engine_bench.rs +++ b/crates/engine-gpu/benches/gpu_engine_bench.rs @@ -6,6 +6,14 @@ use primitive_types::U512; use rand::RngCore; use std::sync::atomic::AtomicBool; +const BENCHMARK_DIFFICULTY: U512 = U512::MAX; + +fn benchmark_context() -> JobContext { + let mut header = [0u8; 32]; + rand::rng().fill_bytes(&mut header); + JobContext::new(header, BENCHMARK_DIFFICULTY) +} + /// Drop thread-local wgpu buffers before `GpuEngine` is dropped. Criterion /// creates a fresh engine per group; without this, TLS buffers outlive the /// device and the next group panics (`Buffer[…] does not exist`). @@ -19,11 +27,11 @@ fn bench_cpu_vs_gpu_small(c: &mut Criterion) { let gpu_engine = GpuEngine::try_new(10_000_000, 0, false).expect("Failed to init GPU"); let cancel_flag = AtomicBool::new(false); let cancel_check = AtomicBoolCancelCheck(&cancel_flag); + let ctx = benchmark_context(); - // Small range: 10K nonces - reasonable for benchmarking let small_range = Range { start: U512::from(0u64), - end: U512::from(10_000u64), + end: U512::from(9_999u64), }; let mut group = c.benchmark_group("small_range_10k"); @@ -32,11 +40,6 @@ fn bench_cpu_vs_gpu_small(c: &mut Criterion) { group.bench_function("cpu", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = cpu_engine.search_range( black_box(&ctx), black_box(small_range.clone()), @@ -48,11 +51,6 @@ fn bench_cpu_vs_gpu_small(c: &mut Criterion) { group.bench_function("gpu", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = gpu_engine.search_range( black_box(&ctx), black_box(small_range.clone()), @@ -71,11 +69,11 @@ fn bench_cpu_vs_gpu_medium(c: &mut Criterion) { let gpu_engine = GpuEngine::try_new(10_000_000, 0, false).expect("Failed to init GPU"); let cancel_flag = AtomicBool::new(false); let cancel_check = AtomicBoolCancelCheck(&cancel_flag); + let ctx = benchmark_context(); - // Medium range: 100K nonces let medium_range = Range { start: U512::from(0u64), - end: U512::from(100_000u64), + end: U512::from(99_999u64), }; let mut group = c.benchmark_group("medium_range_100k"); @@ -84,11 +82,6 @@ fn bench_cpu_vs_gpu_medium(c: &mut Criterion) { group.bench_function("cpu", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = cpu_engine.search_range( black_box(&ctx), black_box(medium_range.clone()), @@ -100,11 +93,6 @@ fn bench_cpu_vs_gpu_medium(c: &mut Criterion) { group.bench_function("gpu", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = gpu_engine.search_range( black_box(&ctx), black_box(medium_range.clone()), @@ -123,11 +111,11 @@ fn bench_cpu_vs_gpu_large(c: &mut Criterion) { let gpu_engine = GpuEngine::try_new(10_000_000, 0, false).expect("Failed to init GPU"); let cancel_flag = AtomicBool::new(false); let cancel_check = AtomicBoolCancelCheck(&cancel_flag); + let ctx = benchmark_context(); - // Large range: 1M nonces - where GPU should really shine let large_range = Range { start: U512::from(0u64), - end: U512::from(1_000_000u64), + end: U512::from(999_999u64), }; let mut group = c.benchmark_group("large_range_1m"); @@ -136,11 +124,6 @@ fn bench_cpu_vs_gpu_large(c: &mut Criterion) { group.bench_function("cpu", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = cpu_engine.search_range( black_box(&ctx), black_box(large_range.clone()), @@ -152,11 +135,6 @@ fn bench_cpu_vs_gpu_large(c: &mut Criterion) { group.bench_function("gpu", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = gpu_engine.search_range( black_box(&ctx), black_box(large_range.clone()), @@ -170,68 +148,16 @@ fn bench_cpu_vs_gpu_large(c: &mut Criterion) { teardown_gpu(gpu_engine); } -fn bench_solution_finding(c: &mut Criterion) { - let cpu_engine = FastCpuEngine::new(10_000); - let gpu_engine = GpuEngine::try_new(10_000_000, 0, false).expect("Failed to init GPU"); - let cancel_flag = AtomicBool::new(false); - let cancel_check = AtomicBoolCancelCheck(&cancel_flag); - - // Range where we expect to find solutions quickly - let solution_range = Range { - start: U512::from(0u64), - end: U512::from(50_000u64), - }; - - let mut group = c.benchmark_group("solution_finding"); - group.sample_size(10); - group.measurement_time(std::time::Duration::from_secs(3)); - - group.bench_function("cpu_find_solution", |b| { - b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(10_000u64); // Easy difficulty - should find solution - let ctx = JobContext::new(header, difficulty); - - let result = cpu_engine.search_range( - black_box(&ctx), - black_box(solution_range.clone()), - black_box(&cancel_check), - ); - black_box(result) - }) - }); - - group.bench_function("gpu_find_solution", |b| { - b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(10_000u64); // Easy difficulty - should find solution - let ctx = JobContext::new(header, difficulty); - - let result = gpu_engine.search_range( - black_box(&ctx), - black_box(solution_range.clone()), - black_box(&cancel_check), - ); - black_box(result) - }) - }); - - group.finish(); - teardown_gpu(gpu_engine); -} - fn bench_throughput_per_second(c: &mut Criterion) { let cpu_engine = FastCpuEngine::new(10_000); let gpu_engine = GpuEngine::try_new(10_000_000, 0, false).expect("Failed to init GPU"); let cancel_flag = AtomicBool::new(false); let cancel_check = AtomicBoolCancelCheck(&cancel_flag); + let ctx = benchmark_context(); - // Fixed time benchmark - see how many hashes we can do in 1 second let throughput_range = Range { start: U512::from(0u64), - end: U512::from(10_000_000u64), // 10M nonce range + end: U512::from(9_999_999u64), }; let mut group = c.benchmark_group("throughput_comparison"); @@ -240,11 +166,6 @@ fn bench_throughput_per_second(c: &mut Criterion) { group.bench_function("cpu_throughput", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = cpu_engine.search_range( black_box(&ctx), black_box(throughput_range.clone()), @@ -256,11 +177,6 @@ fn bench_throughput_per_second(c: &mut Criterion) { group.bench_function("gpu_throughput", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); // High difficulty - no solutions expected - let ctx = JobContext::new(header, difficulty); - let result = gpu_engine.search_range( black_box(&ctx), black_box(throughput_range.clone()), @@ -278,34 +194,29 @@ fn bench_gpu_batch_efficiency(c: &mut Criterion) { let gpu_engine = GpuEngine::try_new(10_000_000, 0, false).expect("Failed to init GPU"); let cancel_flag = AtomicBool::new(false); let cancel_check = AtomicBoolCancelCheck(&cancel_flag); + let ctx = benchmark_context(); let mut group = c.benchmark_group("gpu_batch_sizes"); group.sample_size(10); group.measurement_time(std::time::Duration::from_secs(3)); - // Test different batch sizes to see GPU efficiency let small_batch = Range { start: U512::from(0u64), - end: U512::from(1_000u64), // 1K nonces - very small for GPU + end: U512::from(999u64), }; let medium_batch = Range { start: U512::from(0u64), - end: U512::from(50_000u64), // 50K nonces - medium + end: U512::from(49_999u64), }; let large_batch = Range { start: U512::from(0u64), - end: U512::from(500_000u64), // 500K nonces - large + end: U512::from(499_999u64), }; group.bench_function("gpu_1k_batch", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); - let ctx = JobContext::new(header, difficulty); - let result = gpu_engine.search_range( black_box(&ctx), black_box(small_batch.clone()), @@ -317,11 +228,6 @@ fn bench_gpu_batch_efficiency(c: &mut Criterion) { group.bench_function("gpu_50k_batch", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); - let ctx = JobContext::new(header, difficulty); - let result = gpu_engine.search_range( black_box(&ctx), black_box(medium_batch.clone()), @@ -333,11 +239,6 @@ fn bench_gpu_batch_efficiency(c: &mut Criterion) { group.bench_function("gpu_500k_batch", |b| { b.iter(|| { - let mut header = [0u8; 32]; - rand::rng().fill_bytes(&mut header); - let difficulty = U512::from(u64::MAX); - let ctx = JobContext::new(header, difficulty); - let result = gpu_engine.search_range( black_box(&ctx), black_box(large_batch.clone()), @@ -356,7 +257,6 @@ criterion_group!( bench_cpu_vs_gpu_small, bench_cpu_vs_gpu_medium, bench_cpu_vs_gpu_large, - bench_solution_finding, bench_throughput_per_second, bench_gpu_batch_efficiency ); diff --git a/crates/engine-gpu/examples/hashrate.rs b/crates/engine-gpu/examples/hashrate.rs index cbc3fa1f..6d5edb3b 100644 --- a/crates/engine-gpu/examples/hashrate.rs +++ b/crates/engine-gpu/examples/hashrate.rs @@ -20,7 +20,7 @@ fn main() { let cancel = AtomicBoolCancelCheck(&cancel_flag); let header = [42u8; 32]; - let ctx = engine.prepare_context(header, U512::from(u64::MAX)); + let ctx = engine.prepare_context(header, U512::MAX); let range = Range { start: U512::from(1u64) << 200, diff --git a/crates/miner-cli/src/main.rs b/crates/miner-cli/src/main.rs index c456c204..d68bb511 100644 --- a/crates/miner-cli/src/main.rs +++ b/crates/miner-cli/src/main.rs @@ -13,6 +13,7 @@ use std::time::{Duration, Instant}; const DEFAULT_GPU_BATCH_SIZE: u32 = 1_000_000; const DEFAULT_CUDA_BATCH_SIZE: u32 = 32_000_000; const DEFAULT_CPU_BATCH_SIZE: u64 = 10_000; +const BENCHMARK_DIFFICULTY: U512 = U512::MAX; #[derive(Subcommand, Debug)] enum Command { @@ -396,11 +397,10 @@ async fn run_benchmark( // Random header hash for benchmark let mut header = [0u8; 32]; rand::rng().fill_bytes(&mut header); - let difficulty = U512::MAX; // High difficulty - no solutions expected - let ref_engine = cpu_engine.as_ref().or(gpu_engine.as_ref()).unwrap(); - let ctx = ref_engine.prepare_context(header, difficulty); + let ctx = ref_engine.prepare_context(header, BENCHMARK_DIFFICULTY); + println!("Difficulty: effectively infinite (target 1)"); println!("⛏️ Starting benchmark..."); // Spawn worker threads