From d10703984f9635f2a9fbd3ea36b5eecb7460835e Mon Sep 17 00:00:00 2001 From: rsasaki0109 Date: Wed, 15 Jul 2026 23:14:29 +0900 Subject: [PATCH] perf(vision): accelerate Soft-NMS --- README.md | 17 ++ bench/opencv_comparison/README.md | 3 +- bench/opencv_comparison/manifest.json | 1 + bench/opencv_comparison/test_report.py | 1 + bench/opencv_soft_nms_comparison/README.md | 16 ++ .../opencv_soft_nms_comparison/performance.py | 161 +++++++++++++++ crates/spatialrust-py/src/lib.rs | 12 +- crates/spatialrust-py/tests/test_bindings.py | 3 + .../spatialrust-vision/benches/detection.rs | 97 ++++++++- crates/spatialrust-vision/src/detection.rs | 184 ++++++++++++++---- docs/ROADMAP.md | 1 + docs/site/algorithms.html | 2 +- docs/site/vision2.html | 2 +- ...2026-07-15_soft_nms_opencv_acceleration.md | 73 +++++++ 14 files changed, 533 insertions(+), 40 deletions(-) create mode 100644 bench/opencv_soft_nms_comparison/README.md create mode 100644 bench/opencv_soft_nms_comparison/performance.py create mode 100644 notes/2026-07-15_soft_nms_opencv_acceleration.md diff --git a/README.md b/README.md index 98aac81..052b349 100644 --- a/README.md +++ b/README.md @@ -205,6 +205,23 @@ Both profiles returned exactly the same globally score-ordered indices. See the [batched NMS harness](bench/opencv_batched_nms_comparison/) and dated [receipt](notes/2026-07-15_batched_nms_opencv_acceleration.md). +Soft-NMS retains overlapping detections while decaying their scores. The +linear and Gaussian methods use an active-candidate max scan, cached box areas, +and a non-overlap fast path: + +| Soft-NMS profile | Method | OpenCV | SpatialRust | Result | +| --- | --- | ---: | ---: | ---: | +| 100 candidates | Linear | 0.092 ms | 0.015 ms | **SpatialRust 6.33×** | +| 100 candidates | Gaussian | 0.108 ms | 0.015 ms | **SpatialRust 7.40×** | +| 1,000 candidates | Linear | 5.636 ms | 1.649 ms | **SpatialRust 3.42×** | +| 1,000 candidates | Gaussian | 6.047 ms | 1.293 ms | **SpatialRust 4.68×** | +| 8,400 candidates | Linear | 310.709 ms | 76.660 ms | **SpatialRust 4.05×** | +| 8,400 candidates | Gaussian | 213.696 ms | 39.816 ms | **SpatialRust 5.37×** | + +All profiles exactly matched OpenCV's kept-index order; updated float32 scores +stayed within `1.79e-7`. See the [Soft-NMS harness](bench/opencv_soft_nms_comparison/) +and dated [receipt](notes/2026-07-15_soft_nms_opencv_acceleration.md). + #### Vision accuracy The same deterministic RGB inputs passed all VGA, 1080p, and 4K gates: diff --git a/bench/opencv_comparison/README.md b/bench/opencv_comparison/README.md index d188b9e..a1e5e68 100644 --- a/bench/opencv_comparison/README.md +++ b/bench/opencv_comparison/README.md @@ -32,7 +32,7 @@ VGA, 1080p, and 4K profiles and the initial competitive workload set: 10. colored RGB-D to point cloud 11. AI preprocessing 12. RGB-D to voxel end-to-end -13. detection NMS and class-aware batched NMS post-processing +13. detection NMS, class-aware batched NMS, and Soft-NMS post-processing Exact matches use a JSON `null` PSNR (mathematically infinite) so reports remain strict RFC-compatible JSON. Numerical comparisons retain max/mean/RMS and @@ -52,6 +52,7 @@ then run both current suites: python bench\opencv_comparison\run.py python bench\opencv_nms_comparison\performance.py python bench\opencv_batched_nms_comparison\performance.py +python bench\opencv_soft_nms_comparison\performance.py ``` Reports are written under `target/opencv-comparison/`. Run one suite with diff --git a/bench/opencv_comparison/manifest.json b/bench/opencv_comparison/manifest.json index 3f89177..0d3c9ba 100644 --- a/bench/opencv_comparison/manifest.json +++ b/bench/opencv_comparison/manifest.json @@ -49,6 +49,7 @@ { "id": "ai_preprocess", "domain": "dnn-adapter", "modes": ["allocate", "reuse"] }, { "id": "nms", "domain": "dnn-adapter", "modes": ["postprocess"] }, { "id": "batched_nms", "domain": "dnn-adapter", "modes": ["postprocess"] }, + { "id": "soft_nms", "domain": "dnn-adapter", "modes": ["linear", "gaussian"] }, { "id": "rgbd_to_voxel", "domain": "spatial-e2e", "modes": ["allocate"] } ] } diff --git a/bench/opencv_comparison/test_report.py b/bench/opencv_comparison/test_report.py index f6c019c..01753c0 100644 --- a/bench/opencv_comparison/test_report.py +++ b/bench/opencv_comparison/test_report.py @@ -119,6 +119,7 @@ def test_manifest_reserves_representative_profiles_and_workloads(self) -> None: self.assertIn("ai_preprocess", workloads) self.assertIn("nms", workloads) self.assertIn("batched_nms", workloads) + self.assertIn("soft_nms", workloads) self.assertIn("coefficient_of_variation", statistics) self.assertIn("median_absolute_deviation", statistics) self.assertIn("batch_size", statistics) diff --git a/bench/opencv_soft_nms_comparison/README.md b/bench/opencv_soft_nms_comparison/README.md new file mode 100644 index 0000000..b59ea4d --- /dev/null +++ b/bench/opencv_soft_nms_comparison/README.md @@ -0,0 +1,16 @@ +# OpenCV Soft-NMS comparison + +This harness compares SpatialRust `soft_nms` with OpenCV `dnn.softNMSBoxes` +for linear and Gaussian score decay. Both receive the same deterministic +integer-coordinate boxes, float32 scores, score/IoU thresholds, and sigma. +Kept indices must match exactly and updated scores must remain within `4e-7` +absolute error before timings are published. + +```powershell +python bench/opencv_soft_nms_comparison/performance.py ` + --output target/opencv-soft-nms-performance.json +``` + +The report follows `spatialrust.opencv-comparison.v1` and records raw samples, +dispersion, library versions, thread policy, and the host environment. Results +are machine-specific and must not be generalized beyond the named workload. diff --git a/bench/opencv_soft_nms_comparison/performance.py b/bench/opencv_soft_nms_comparison/performance.py new file mode 100644 index 0000000..413e6c6 --- /dev/null +++ b/bench/opencv_soft_nms_comparison/performance.py @@ -0,0 +1,161 @@ +"""Reproducible linear and Gaussian Soft-NMS comparison with OpenCV.""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +import cv2 +import numpy as np +import spatialrust as sr + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from opencv_comparison.report import emit_report, environment, make_report, timed_pair + + +PROFILES = { + "small_100": (100, 50), + "medium_1000": (1_000, 20), + "yolo_8400": (8_400, 8), +} +METHODS = { + "linear": cv2.dnn.SOFT_NMSMETHOD_SOFTNMS_LINEAR, + "gaussian": cv2.dnn.SOFT_NMSMETHOD_SOFTNMS_GAUSSIAN, +} +SCORE_TOLERANCE = 4.0e-7 + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("--output", type=Path) + parser.add_argument("--profiles", default=",".join(PROFILES)) + parser.add_argument("--methods", default=",".join(METHODS)) + parser.add_argument("--warmup", type=int, default=3) + return parser.parse_args() + + +def main() -> None: + args = parse_args() + selected_profiles = [name.strip() for name in args.profiles.split(",") if name.strip()] + selected_methods = [name.strip() for name in args.methods.split(",") if name.strip()] + unknown_profiles = sorted(set(selected_profiles) - PROFILES.keys()) + unknown_methods = sorted(set(selected_methods) - METHODS.keys()) + if unknown_profiles: + raise ValueError(f"unknown profiles: {', '.join(unknown_profiles)}") + if unknown_methods: + raise ValueError(f"unknown methods: {', '.join(unknown_methods)}") + if args.warmup < 0: + raise ValueError("warmup must be non-negative") + if not hasattr(cv2.dnn, "softNMSBoxes"): + raise RuntimeError("OpenCV build does not expose dnn.softNMSBoxes") + if hasattr(cv2, "ocl"): + cv2.ocl.setUseOpenCL(False) + + rng = np.random.default_rng(129) + results: dict[str, object] = {} + for profile in selected_profiles: + count, repeats = PROFILES[profile] + origins = rng.integers(0, 600, size=(count, 2), dtype=np.int32) + sizes = rng.integers(5, 100, size=(count, 2), dtype=np.int32) + boxes_xywh = np.column_stack((origins, sizes)).astype(np.int32) + boxes_xyxy = np.column_stack((origins, origins + sizes)).astype(np.float32) + scores = rng.random(count, dtype=np.float32) + profile_results: dict[str, object] = {} + + for method_name in selected_methods: + opencv_method = METHODS[method_name] + + def opencv_soft_nms() -> tuple[np.ndarray, np.ndarray]: + return cv2.dnn.softNMSBoxes( + boxes_xywh, + scores, + 0.25, + 0.5, + 0, + 0.5, + opencv_method, + ) + + def spatialrust_soft_nms() -> tuple[list[int], list[float]]: + return sr.soft_nms(boxes_xyxy, scores, 0.25, 0.5, method_name, 0.5) + + expected_scores, expected_indices = opencv_soft_nms() + actual_indices, actual_scores = spatialrust_soft_nms() + actual_indices_array = np.asarray(actual_indices, dtype=np.int64) + actual_scores_array = np.asarray(actual_scores, dtype=np.float32) + indices_exact = bool(np.array_equal(expected_indices, actual_indices_array)) + score_max_error = float( + np.max(np.abs(expected_scores - actual_scores_array)) + ) if expected_scores.size else 0.0 + if not indices_exact: + raise AssertionError(f"{profile}/{method_name} Soft-NMS index mismatch") + if score_max_error > SCORE_TOLERANCE: + raise AssertionError( + f"{profile}/{method_name} score error {score_max_error} " + f"> {SCORE_TOLERANCE}" + ) + + _, _, opencv_timing, spatialrust_timing = timed_pair( + opencv_soft_nms, + spatialrust_soft_nms, + warmup=args.warmup, + repeats=repeats, + seed=131, + min_sample_time_ms=5.0, + ) + opencv_ms = float(opencv_timing["median"]) + spatialrust_ms = float(spatialrust_timing["median"]) + profile_results[method_name] = { + "kept_count": len(actual_indices), + "indices_exact": indices_exact, + "score_max_absolute_error": score_max_error, + "score_tolerance": SCORE_TOLERANCE, + "opencv": opencv_timing, + "spatialrust": spatialrust_timing, + "spatialrust_speedup": opencv_ms / spatialrust_ms, + "faster_implementation": ( + "spatialrust" if spatialrust_ms < opencv_ms else "opencv" + ), + } + results[profile] = { + "box_count": count, + "score_threshold": 0.25, + "iou_threshold": 0.5, + "sigma": 0.5, + "methods": profile_results, + } + + receipt = environment( + opencv_version=cv2.__version__, spatialrust_version=sr.__version__ + ) + receipt["opencv_threads"] = cv2.getNumThreads() + receipt["opencv_opencl_enabled"] = bool( + hasattr(cv2, "ocl") and cv2.ocl.useOpenCL() + ) + report = make_report( + suite="opencv-soft-nms-performance", + kind="performance", + status="pass", + environment_receipt=receipt, + results={ + "methodology": { + "timing_scope": "Python API call returning updated scores and kept indices", + "paired_interleaved": True, + "input_seed": 129, + "random_order_seed": 131, + "minimum_sample_time_ms": 5.0, + "box_format": { + "opencv": "xywh int32 NumPy array", + "spatialrust": "corresponding xyxy float32 NumPy array", + }, + "thread_policy": "library defaults; OpenCV thread count recorded", + }, + "profiles": results, + }, + ) + emit_report(report, args.output) + + +if __name__ == "__main__": + main() diff --git a/crates/spatialrust-py/src/lib.rs b/crates/spatialrust-py/src/lib.rs index 5c3b180..f734efc 100644 --- a/crates/spatialrust-py/src/lib.rs +++ b/crates/spatialrust-py/src/lib.rs @@ -3134,8 +3134,16 @@ fn soft_nms( "gaussian" => SoftNmsMethod::Gaussian { sigma }, other => return Err(PyValueError::new_err(format!("unknown Soft-NMS method `{other}`"))), }; - let scores: Vec = scores.as_array().iter().copied().collect(); - let result = soft_nms_op(&native_boxes, &scores, score_threshold, iou_threshold, method) + let scores_view = scores.as_array(); + let packed_scores; + let scores = match scores_view.as_slice() { + Some(scores) => scores, + None => { + packed_scores = scores_view.iter().copied().collect::>(); + packed_scores.as_slice() + } + }; + let result = soft_nms_op(&native_boxes, scores, score_threshold, iou_threshold, method) .map_err(to_py_err)?; Ok(( result.iter().map(|value| value.index).collect(), diff --git a/crates/spatialrust-py/tests/test_bindings.py b/crates/spatialrust-py/tests/test_bindings.py index 759a300..7604164 100644 --- a/crates/spatialrust-py/tests/test_bindings.py +++ b/crates/spatialrust-py/tests/test_bindings.py @@ -193,6 +193,9 @@ def test_detection_nms_and_soft_nms(): assert indices[0] == 0 assert len(indices) == len(updated) == 3 assert updated[-1] < 0.8 + view_indices, view_updated = sr.soft_nms(boxes, score_storage[::2], method="linear") + assert view_indices == indices + np.testing.assert_array_equal(view_updated, updated) def test_mask_components_contours_and_rle(): diff --git a/crates/spatialrust-vision/benches/detection.rs b/crates/spatialrust-vision/benches/detection.rs index 3cbd82e..b0a725b 100644 --- a/crates/spatialrust-vision/benches/detection.rs +++ b/crates/spatialrust-vision/benches/detection.rs @@ -1,5 +1,6 @@ use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; -use spatialrust_vision::{batched_nms, nms, BoundingBox2, Detection}; +use spatialrust_vision::{batched_nms, nms, soft_nms, BoundingBox2, Detection, SoftNmsMethod}; +use std::cmp::Ordering; fn benchmark_nms(c: &mut Criterion) { let mut group = c.benchmark_group("nms_xyxy_f32"); @@ -38,6 +39,100 @@ fn benchmark_nms(c: &mut Criterion) { }); } group.finish(); + + let mut group = c.benchmark_group("soft_nms_linear_xyxy_f32"); + group.sample_size(10); + for &count in &[100_usize, 1_000, 8_400] { + let (boxes, scores) = detections(count); + group.throughput(Throughput::Elements(count as u64)); + group.bench_function(BenchmarkId::from_parameter(count), |b| { + b.iter(|| { + black_box( + soft_nms( + black_box(&boxes), + black_box(&scores), + black_box(0.25), + black_box(0.5), + black_box(SoftNmsMethod::Linear), + ) + .unwrap(), + ) + }); + }); + group.bench_function(BenchmarkId::new("sorting_baseline", count), |b| { + b.iter(|| { + black_box(soft_nms_sorting_baseline( + black_box(&boxes), + black_box(&scores), + black_box(0.25), + black_box(0.5), + )) + }); + }); + } + group.finish(); +} + +#[derive(Clone, Copy)] +struct BaselineCandidate { + index: usize, + score: f32, +} + +fn soft_nms_sorting_baseline( + boxes: &[BoundingBox2], + scores: &[f32], + score_threshold: f32, + iou_threshold: f32, +) -> Vec { + let mut candidates = scores + .iter() + .copied() + .enumerate() + .map(|(index, score)| BaselineCandidate { index, score }) + .collect::>(); + let areas = boxes.iter().copied().map(BoundingBox2::area).collect::>(); + let mut output = Vec::with_capacity(candidates.len()); + while !candidates.is_empty() { + candidates.sort_by(|left, right| { + right + .score + .partial_cmp(&left.score) + .unwrap_or(Ordering::Equal) + .then_with(|| left.index.cmp(&right.index)) + }); + let selected = candidates.remove(0); + if selected.score < score_threshold { + break; + } + output.push(selected); + for candidate in &mut candidates { + let overlap = cached_iou( + boxes[selected.index], + areas[selected.index], + boxes[candidate.index], + areas[candidate.index], + ); + if overlap > iou_threshold { + candidate.score *= 1.0 - overlap; + } + } + candidates.retain(|candidate| candidate.score >= score_threshold); + } + output +} + +fn cached_iou(left: BoundingBox2, left_area: f32, right: BoundingBox2, right_area: f32) -> f32 { + let width = left.x_max.min(right.x_max) - left.x_min.max(right.x_min); + if width <= 0.0 { + return 0.0; + } + let height = left.y_max.min(right.y_max) - left.y_min.max(right.y_min); + if height <= 0.0 { + return 0.0; + } + let intersection = width * height; + intersection / (left_area + right_area - intersection) } fn detections(count: usize) -> (Vec, Vec) { diff --git a/crates/spatialrust-vision/src/detection.rs b/crates/spatialrust-vision/src/detection.rs index efc9e31..da81437 100644 --- a/crates/spatialrust-vision/src/detection.rs +++ b/crates/spatialrust-vision/src/detection.rs @@ -259,48 +259,87 @@ pub fn soft_nms( .map(|(index, score)| ScoredIndex { index, score }) .collect(); let areas: Vec = boxes.iter().copied().map(BoundingBox2::area).collect(); - let mut output = Vec::new(); - while !candidates.is_empty() { - candidates.sort_by(score_order); - let selected = candidates.remove(0); - if selected.score < score_threshold { + let mut output = Vec::with_capacity(candidates.len()); + let mut best = best_scored_candidate(&candidates); + while let Some(best_index) = best { + if candidates[best_index].score < score_threshold { break; } + let selected = candidates.swap_remove(best_index); output.push(selected); - for candidate in &mut candidates { + let selected_box = boxes[selected.index]; + let selected_area = areas[selected.index]; + let mut active_count = 0; + let mut next_best = None; + for read_index in 0..candidates.len() { + let mut candidate = candidates[read_index]; let overlap = iou_with_areas( - boxes[selected.index], - areas[selected.index], + selected_box, + selected_area, boxes[candidate.index], areas[candidate.index], ); - let weight = match method { - SoftNmsMethod::Hard => { - if overlap <= iou_threshold { - 1.0 - } else { - 0.0 + if overlap != 0.0 { + match method { + SoftNmsMethod::Hard => { + if overlap > iou_threshold { + candidate.score *= 0.0; + } } - } - SoftNmsMethod::Linear => { - if overlap > iou_threshold { - 1.0 - overlap - } else { - 1.0 + SoftNmsMethod::Linear => { + if overlap > iou_threshold { + candidate.score *= 1.0 - overlap; + } + } + SoftNmsMethod::Gaussian { sigma } => { + candidate.score *= (-(overlap * overlap) / sigma).exp(); } } - SoftNmsMethod::Gaussian { sigma } => (-(overlap * overlap) / sigma).exp(), - }; - candidate.score *= weight; + } + if candidate.score >= score_threshold { + candidates[active_count] = candidate; + let is_next_best = match next_best { + Some(index) => scored_candidate_precedes(candidate, candidates[index]), + None => true, + }; + if is_next_best { + next_best = Some(active_count); + } + active_count += 1; + } } - candidates.retain(|candidate| candidate.score >= score_threshold); + candidates.truncate(active_count); + best = next_best; } Ok(output) } +fn best_scored_candidate(candidates: &[ScoredIndex]) -> Option { + if candidates.is_empty() { + return None; + } + let mut best = 0; + for index in 1..candidates.len() { + if scored_candidate_precedes(candidates[index], candidates[best]) { + best = index; + } + } + Some(best) +} + +fn scored_candidate_precedes(left: ScoredIndex, right: ScoredIndex) -> bool { + left.score > right.score || left.score == right.score && left.index < right.index +} + fn iou_with_areas(left: BoundingBox2, left_area: f32, right: BoundingBox2, right_area: f32) -> f32 { - let intersection_width = (left.x_max.min(right.x_max) - left.x_min.max(right.x_min)).max(0.0); - let intersection_height = (left.y_max.min(right.y_max) - left.y_min.max(right.y_min)).max(0.0); + let intersection_width = left.x_max.min(right.x_max) - left.x_min.max(right.x_min); + if intersection_width <= 0.0 { + return 0.0; + } + let intersection_height = left.y_max.min(right.y_max) - left.y_min.max(right.y_min); + if intersection_height <= 0.0 { + return 0.0; + } let intersection = intersection_width * intersection_height; let union = left_area + right_area - intersection; if union > 0.0 { @@ -346,14 +385,6 @@ fn sort_indices_by_score(indices: &mut [usize], scores: &[f32]) { }); } -fn score_order(left: &ScoredIndex, right: &ScoredIndex) -> Ordering { - right - .score - .partial_cmp(&left.score) - .unwrap_or(Ordering::Equal) - .then_with(|| left.index.cmp(&right.index)) -} - #[cfg(test)] mod tests { use super::{ @@ -470,5 +501,90 @@ mod tests { let result = soft_nms(&boxes, &[0.9, 0.8], 0.01, 0.5, SoftNmsMethod::Linear).unwrap(); assert_eq!(result[0].index, 0); assert!(result[1].score < 0.8); + assert!(soft_nms(&boxes, &[0.9, 0.8], 0.01, 0.5, SoftNmsMethod::Gaussian { sigma: 0.0 },) + .is_err()); + } + + #[test] + fn soft_nms_selection_scan_preserves_sorting_semantics() { + let mut state = 127_u64; + let boxes = (0..73) + .map(|_| { + let x = sample(&mut state) * 64.0; + let y = sample(&mut state) * 64.0; + let width = 1.0 + sample(&mut state) * 20.0; + let height = 1.0 + sample(&mut state) * 20.0; + bbox(x, y, x + width, y + height) + }) + .collect::>(); + let mut scores = (0..boxes.len()).map(|_| sample(&mut state)).collect::>(); + scores[5] = scores[2]; + for method in + [SoftNmsMethod::Hard, SoftNmsMethod::Linear, SoftNmsMethod::Gaussian { sigma: 0.5 }] + { + let actual = soft_nms(&boxes, &scores, 0.2, 0.45, method).unwrap(); + let expected = sorting_soft_nms_reference(&boxes, &scores, 0.2, 0.45, method); + assert_eq!(actual, expected); + } + + let overlapping = [bbox(0.0, 0.0, 2.0, 2.0), bbox(0.0, 0.0, 2.0, 2.0)]; + let negative_scores = [1.0, -0.6]; + assert_eq!( + soft_nms(&overlapping, &negative_scores, -0.5, 0.45, SoftNmsMethod::Hard).unwrap(), + sorting_soft_nms_reference( + &overlapping, + &negative_scores, + -0.5, + 0.45, + SoftNmsMethod::Hard, + ), + ); + } + + fn sorting_soft_nms_reference( + boxes: &[BoundingBox2], + scores: &[f32], + score_threshold: f32, + iou_threshold: f32, + method: SoftNmsMethod, + ) -> Vec { + let mut candidates = scores + .iter() + .copied() + .enumerate() + .map(|(index, score)| super::ScoredIndex { index, score }) + .collect::>(); + let mut output = Vec::new(); + while !candidates.is_empty() { + candidates.sort_by(|left, right| { + right + .score + .partial_cmp(&left.score) + .unwrap() + .then_with(|| left.index.cmp(&right.index)) + }); + let selected = candidates.remove(0); + if selected.score < score_threshold { + break; + } + output.push(selected); + for candidate in &mut candidates { + let overlap = boxes[selected.index].iou(boxes[candidate.index]); + let weight = match method { + SoftNmsMethod::Hard => (overlap <= iou_threshold) as u8 as f32, + SoftNmsMethod::Linear => { + if overlap > iou_threshold { + 1.0 - overlap + } else { + 1.0 + } + } + SoftNmsMethod::Gaussian { sigma } => (-(overlap * overlap) / sigma).exp(), + }; + candidate.score *= weight; + } + candidates.retain(|candidate| candidate.score >= score_threshold); + } + output } } diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 7226c3b..5ef0a45 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -620,6 +620,7 @@ to one implicitly, and GPU receipts must retain named upload/readback stages. | 114F | Complete | Cache EDT parabola heights and balance column tasks for dense masks | exact OpenCV parity and 4K Python reuse win | | 114G | Complete | Cache NMS box geometry and avoid packed Python score copies | exact OpenCV index parity and 100/1,000/8,400-candidate wins | | 114H | Complete | Bucket class-aware NMS keeps and expose one-call Python batched NMS | exact OpenCV parity and 26.38×/97.25× wins | +| 114I | Complete | One-pass active-set Soft-NMS selection, disjoint IoU exit, and borrowed Python scores | exact indices, bounded scores, and 3.42×–7.40× wins | ### Epic 115 delivery slices diff --git a/docs/site/algorithms.html b/docs/site/algorithms.html index babf0f0..1adc840 100644 --- a/docs/site/algorithms.html +++ b/docs/site/algorithms.html @@ -36,7 +36,7 @@

Algorithm catalog

Image analysisFixed/Otsu/adaptive threshold, histogram, equalization, CLAHE, integral image, Cannyspatialrust-vision · imgproc-analysis, imgproc-cannyCPU Local featuresFAST, Harris, Shi–Tomasi, ORB, descriptor matching, grid selection, pyramidal Lucas–Kanade trackingspatialrust-vision · feature2dCPU Dense visionExact Euclidean distance transform with reusable workspace/output, connected components, contours, polygon approximation, mask RLE, depth/confidence/flow/point mapsspatialrust-vision · denseCPU parallel - Detection post-processingIoU/GIoU, greedy NMS, class-aware batched NMS, and hard/linear/Gaussian Soft-NMS. Seeded Python NMS is 3.22×–8.95× faster than OpenCV; class-aware batched NMS is 26.38×–97.25× faster on the recorded 1,000/8,400-candidate workloads. Both require exact kept-index parity. NMS harness · batched harness.spatialrust-vision · detectionCPU + Detection post-processingIoU/GIoU, greedy NMS, class-aware batched NMS, and hard/linear/Gaussian Soft-NMS. Seeded Python NMS is 3.22×–8.95× faster than OpenCV; batched NMS is 26.38×–97.25× faster; linear/Gaussian Soft-NMS is 3.42×–7.40× faster. NMS indices are exact and Soft-NMS scores stay within 1.79e-7. NMS · batched · Soft-NMS.spatialrust-vision · detectionCPU Multiview geometryHomography, fundamental/essential matrices, RANSAC, triangulation, relative pose, PnP/PnP-RANSACspatialrust-vision · geometryCPU Stereo and odometryStereo rectification, block matching, disparity-to-depth/XYZ, monocular and RGB-D visual odometryspatialrust-vision · geometry, odometryCPU VideoDense block flow, adaptive background model, multi-object tracker, timestamped pull sourcesspatialrust-vision · videoCPU diff --git a/docs/site/vision2.html b/docs/site/vision2.html index 10e081e..b6ccfdd 100644 --- a/docs/site/vision2.html +++ b/docs/site/vision2.html @@ -51,7 +51,7 @@

Latest measured outcome

Exact EDT at 4K

With caller-owned output and workspace reuse, SpatialRust measured 40.66 ms versus OpenCV 43.33 ms: a 1.07× lead on the recorded Windows host.

Accuracy preserved

The VGA, 1080p, and 4K canonical masks retain maximum absolute error 0.0 against OpenCV's precise L2 distance transform.

-

NMS post-processing

SpatialRust measured 3.22×–8.95× faster for class-agnostic NMS and 26.38×–97.25× faster for class-aware batched NMS, with kept indices exactly matching OpenCV on every recorded profile.

+

NMS post-processing

SpatialRust measured 3.22×–8.95× faster for NMS, 26.38×–97.25× for batched NMS, and 3.42×–7.40× for linear/Gaussian Soft-NMS. Indices exactly match OpenCV; Soft-NMS scores stay within 1.79e-7.

This is a workload- and host-specific result. VGA and 1080p reuse remain narrow OpenCV wins; the repository receipt contains the reproducible methodology.

diff --git a/notes/2026-07-15_soft_nms_opencv_acceleration.md b/notes/2026-07-15_soft_nms_opencv_acceleration.md new file mode 100644 index 0000000..247bcea --- /dev/null +++ b/notes/2026-07-15_soft_nms_opencv_acceleration.md @@ -0,0 +1,73 @@ +# Soft-NMS OpenCV acceleration receipt — 2026-07-15 + +## Outcome + +SpatialRust linear and Gaussian Soft-NMS now maintain only active candidates. +One traversal decays scores, compacts survivors, and selects the next maximum; +the chosen item is removed in constant time. Cached box areas are reused, +disjoint boxes return zero IoU before division, and no-op score updates are +skipped. The Python binding borrows contiguous float32 scores and only packs +non-contiguous views. + +The public safe Rust API and Python signature remain compatible. Hard, +linear, and Gaussian methods preserve deterministic descending-score order, +including original-index tie breaking. + +Soft-NMS is useful for crowded detection scenes because it reduces scores as +a function of overlap rather than deleting every overlapping detection. The +original paper reports improvements without retraining and the same quadratic +complexity class as greedy NMS. + +## Correctness contract + +The oracle is OpenCV +[`dnn.softNMSBoxes`](https://docs.opencv.org/master/df/d57/namespacecv_1_1dnn.html). +Both implementations receive the same deterministic integer-coordinate boxes, +float32 scores, score threshold `0.25`, IoU threshold `0.5`, and Gaussian sigma +`0.5`. OpenCV receives `xywh`; SpatialRust receives the corresponding `xyxy`. + +Across 100, 1,000, and 8,400 candidates for both linear and Gaussian decay: + +- kept indices and ordering matched exactly; +- maximum updated-score error was `1.7881393e-7`; +- all results passed the declared `4e-7` absolute score tolerance. + +An additional 48 randomized linear/Gaussian cases retained exact indices and +the same observed maximum score error. Rust tests compare the optimized active +set against the previous full-sort semantics for Hard, Linear, and Gaussian, +including tied scores. Python tests cover contiguous and non-contiguous score +arrays. + +## Performance + +Host: Windows 11, Intel 6-core/12-thread CPU, CPython 3.12.10, OpenCV 4.10, +OpenCL disabled. Timings are seeded, randomized, interleaved Python API medians +after three warmups, batch short calls to at least 5 ms, and include both +returned score and index collections. + +| Candidates | Method | Repeats | OpenCV | SpatialRust | SpatialRust speedup | +| ---: | --- | ---: | ---: | ---: | ---: | +| 100 | Linear | 50 | 0.0922 ms | 0.0146 ms | **6.33×** | +| 100 | Gaussian | 50 | 0.1084 ms | 0.0146 ms | **7.40×** | +| 1,000 | Linear | 20 | 5.6359 ms | 1.6494 ms | **3.42×** | +| 1,000 | Gaussian | 20 | 6.0473 ms | 1.2928 ms | **4.68×** | +| 8,400 | Linear | 8 | 310.7088 ms | 76.6595 ms | **4.05×** | +| 8,400 | Gaussian | 8 | 213.6960 ms | 39.8155 ms | **5.37×** | + +Native Criterion compares the one-pass active set with a retained full-sort +baseline on the same separately seeded linear workload: + +| Candidates | One-pass active set | Sorting baseline | Native improvement | +| ---: | ---: | ---: | ---: | +| 100 | 22.1 µs | 27.3 µs | **1.24×** | +| 1,000 | 2.14 ms | 2.18 ms | **1.02×** | +| 8,400 | 88.7 ms | 108.2 ms | **1.22×** | + +These are scoped host/workload measurements, not universal claims. + +Reproduce the machine-readable report with: + +```powershell +python bench/opencv_soft_nms_comparison/performance.py ` + --output target/opencv-soft-nms-performance.json +```