Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -205,6 +205,23 @@ Both profiles returned exactly the same globally score-ordered indices. See
the [batched NMS harness](bench/opencv_batched_nms_comparison/) and dated
[receipt](notes/2026-07-15_batched_nms_opencv_acceleration.md).

Soft-NMS retains overlapping detections while decaying their scores. The
linear and Gaussian methods use an active-candidate max scan, cached box areas,
and a non-overlap fast path:

| Soft-NMS profile | Method | OpenCV | SpatialRust | Result |
| --- | --- | ---: | ---: | ---: |
| 100 candidates | Linear | 0.092 ms | 0.015 ms | **SpatialRust 6.33×** |
| 100 candidates | Gaussian | 0.108 ms | 0.015 ms | **SpatialRust 7.40×** |
| 1,000 candidates | Linear | 5.636 ms | 1.649 ms | **SpatialRust 3.42×** |
| 1,000 candidates | Gaussian | 6.047 ms | 1.293 ms | **SpatialRust 4.68×** |
| 8,400 candidates | Linear | 310.709 ms | 76.660 ms | **SpatialRust 4.05×** |
| 8,400 candidates | Gaussian | 213.696 ms | 39.816 ms | **SpatialRust 5.37×** |

All profiles exactly matched OpenCV's kept-index order; updated float32 scores
stayed within `1.79e-7`. See the [Soft-NMS harness](bench/opencv_soft_nms_comparison/)
and dated [receipt](notes/2026-07-15_soft_nms_opencv_acceleration.md).

#### Vision accuracy

The same deterministic RGB inputs passed all VGA, 1080p, and 4K gates:
Expand Down
3 changes: 2 additions & 1 deletion bench/opencv_comparison/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,7 @@ VGA, 1080p, and 4K profiles and the initial competitive workload set:
10. colored RGB-D to point cloud
11. AI preprocessing
12. RGB-D to voxel end-to-end
13. detection NMS and class-aware batched NMS post-processing
13. detection NMS, class-aware batched NMS, and Soft-NMS post-processing

Exact matches use a JSON `null` PSNR (mathematically infinite) so reports remain
strict RFC-compatible JSON. Numerical comparisons retain max/mean/RMS and
Expand All @@ -52,6 +52,7 @@ then run both current suites:
python bench\opencv_comparison\run.py
python bench\opencv_nms_comparison\performance.py
python bench\opencv_batched_nms_comparison\performance.py
python bench\opencv_soft_nms_comparison\performance.py
```

Reports are written under `target/opencv-comparison/`. Run one suite with
Expand Down
1 change: 1 addition & 0 deletions bench/opencv_comparison/manifest.json
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,7 @@
{ "id": "ai_preprocess", "domain": "dnn-adapter", "modes": ["allocate", "reuse"] },
{ "id": "nms", "domain": "dnn-adapter", "modes": ["postprocess"] },
{ "id": "batched_nms", "domain": "dnn-adapter", "modes": ["postprocess"] },
{ "id": "soft_nms", "domain": "dnn-adapter", "modes": ["linear", "gaussian"] },
{ "id": "rgbd_to_voxel", "domain": "spatial-e2e", "modes": ["allocate"] }
]
}
1 change: 1 addition & 0 deletions bench/opencv_comparison/test_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -119,6 +119,7 @@ def test_manifest_reserves_representative_profiles_and_workloads(self) -> None:
self.assertIn("ai_preprocess", workloads)
self.assertIn("nms", workloads)
self.assertIn("batched_nms", workloads)
self.assertIn("soft_nms", workloads)
self.assertIn("coefficient_of_variation", statistics)
self.assertIn("median_absolute_deviation", statistics)
self.assertIn("batch_size", statistics)
Expand Down
16 changes: 16 additions & 0 deletions bench/opencv_soft_nms_comparison/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
# OpenCV Soft-NMS comparison

This harness compares SpatialRust `soft_nms` with OpenCV `dnn.softNMSBoxes`
for linear and Gaussian score decay. Both receive the same deterministic
integer-coordinate boxes, float32 scores, score/IoU thresholds, and sigma.
Kept indices must match exactly and updated scores must remain within `4e-7`
absolute error before timings are published.

```powershell
python bench/opencv_soft_nms_comparison/performance.py `
--output target/opencv-soft-nms-performance.json
```

The report follows `spatialrust.opencv-comparison.v1` and records raw samples,
dispersion, library versions, thread policy, and the host environment. Results
are machine-specific and must not be generalized beyond the named workload.
161 changes: 161 additions & 0 deletions bench/opencv_soft_nms_comparison/performance.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,161 @@
"""Reproducible linear and Gaussian Soft-NMS comparison with OpenCV."""

from __future__ import annotations

import argparse
import sys
from pathlib import Path

import cv2
import numpy as np
import spatialrust as sr

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from opencv_comparison.report import emit_report, environment, make_report, timed_pair


PROFILES = {
"small_100": (100, 50),
"medium_1000": (1_000, 20),
"yolo_8400": (8_400, 8),
}
METHODS = {
"linear": cv2.dnn.SOFT_NMSMETHOD_SOFTNMS_LINEAR,
"gaussian": cv2.dnn.SOFT_NMSMETHOD_SOFTNMS_GAUSSIAN,
}
SCORE_TOLERANCE = 4.0e-7


def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--output", type=Path)
parser.add_argument("--profiles", default=",".join(PROFILES))
parser.add_argument("--methods", default=",".join(METHODS))
parser.add_argument("--warmup", type=int, default=3)
return parser.parse_args()


def main() -> None:
args = parse_args()
selected_profiles = [name.strip() for name in args.profiles.split(",") if name.strip()]
selected_methods = [name.strip() for name in args.methods.split(",") if name.strip()]
unknown_profiles = sorted(set(selected_profiles) - PROFILES.keys())
unknown_methods = sorted(set(selected_methods) - METHODS.keys())
if unknown_profiles:
raise ValueError(f"unknown profiles: {', '.join(unknown_profiles)}")
if unknown_methods:
raise ValueError(f"unknown methods: {', '.join(unknown_methods)}")
if args.warmup < 0:
raise ValueError("warmup must be non-negative")
if not hasattr(cv2.dnn, "softNMSBoxes"):
raise RuntimeError("OpenCV build does not expose dnn.softNMSBoxes")
if hasattr(cv2, "ocl"):
cv2.ocl.setUseOpenCL(False)

rng = np.random.default_rng(129)
results: dict[str, object] = {}
for profile in selected_profiles:
count, repeats = PROFILES[profile]
origins = rng.integers(0, 600, size=(count, 2), dtype=np.int32)
sizes = rng.integers(5, 100, size=(count, 2), dtype=np.int32)
boxes_xywh = np.column_stack((origins, sizes)).astype(np.int32)
boxes_xyxy = np.column_stack((origins, origins + sizes)).astype(np.float32)
scores = rng.random(count, dtype=np.float32)
profile_results: dict[str, object] = {}

for method_name in selected_methods:
opencv_method = METHODS[method_name]

def opencv_soft_nms() -> tuple[np.ndarray, np.ndarray]:
return cv2.dnn.softNMSBoxes(
boxes_xywh,
scores,
0.25,
0.5,
0,
0.5,
opencv_method,
)

def spatialrust_soft_nms() -> tuple[list[int], list[float]]:
return sr.soft_nms(boxes_xyxy, scores, 0.25, 0.5, method_name, 0.5)

expected_scores, expected_indices = opencv_soft_nms()
actual_indices, actual_scores = spatialrust_soft_nms()
actual_indices_array = np.asarray(actual_indices, dtype=np.int64)
actual_scores_array = np.asarray(actual_scores, dtype=np.float32)
indices_exact = bool(np.array_equal(expected_indices, actual_indices_array))
score_max_error = float(
np.max(np.abs(expected_scores - actual_scores_array))
) if expected_scores.size else 0.0
if not indices_exact:
raise AssertionError(f"{profile}/{method_name} Soft-NMS index mismatch")
if score_max_error > SCORE_TOLERANCE:
raise AssertionError(
f"{profile}/{method_name} score error {score_max_error} "
f"> {SCORE_TOLERANCE}"
)

_, _, opencv_timing, spatialrust_timing = timed_pair(
opencv_soft_nms,
spatialrust_soft_nms,
warmup=args.warmup,
repeats=repeats,
seed=131,
min_sample_time_ms=5.0,
)
opencv_ms = float(opencv_timing["median"])
spatialrust_ms = float(spatialrust_timing["median"])
profile_results[method_name] = {
"kept_count": len(actual_indices),
"indices_exact": indices_exact,
"score_max_absolute_error": score_max_error,
"score_tolerance": SCORE_TOLERANCE,
"opencv": opencv_timing,
"spatialrust": spatialrust_timing,
"spatialrust_speedup": opencv_ms / spatialrust_ms,
"faster_implementation": (
"spatialrust" if spatialrust_ms < opencv_ms else "opencv"
),
}
results[profile] = {
"box_count": count,
"score_threshold": 0.25,
"iou_threshold": 0.5,
"sigma": 0.5,
"methods": profile_results,
}

receipt = environment(
opencv_version=cv2.__version__, spatialrust_version=sr.__version__
)
receipt["opencv_threads"] = cv2.getNumThreads()
receipt["opencv_opencl_enabled"] = bool(
hasattr(cv2, "ocl") and cv2.ocl.useOpenCL()
)
report = make_report(
suite="opencv-soft-nms-performance",
kind="performance",
status="pass",
environment_receipt=receipt,
results={
"methodology": {
"timing_scope": "Python API call returning updated scores and kept indices",
"paired_interleaved": True,
"input_seed": 129,
"random_order_seed": 131,
"minimum_sample_time_ms": 5.0,
"box_format": {
"opencv": "xywh int32 NumPy array",
"spatialrust": "corresponding xyxy float32 NumPy array",
},
"thread_policy": "library defaults; OpenCV thread count recorded",
},
"profiles": results,
},
)
emit_report(report, args.output)


if __name__ == "__main__":
main()
12 changes: 10 additions & 2 deletions crates/spatialrust-py/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3134,8 +3134,16 @@ fn soft_nms(
"gaussian" => SoftNmsMethod::Gaussian { sigma },
other => return Err(PyValueError::new_err(format!("unknown Soft-NMS method `{other}`"))),
};
let scores: Vec<f32> = scores.as_array().iter().copied().collect();
let result = soft_nms_op(&native_boxes, &scores, score_threshold, iou_threshold, method)
let scores_view = scores.as_array();
let packed_scores;
let scores = match scores_view.as_slice() {
Some(scores) => scores,
None => {
packed_scores = scores_view.iter().copied().collect::<Vec<_>>();
packed_scores.as_slice()
}
};
let result = soft_nms_op(&native_boxes, scores, score_threshold, iou_threshold, method)
.map_err(to_py_err)?;
Ok((
result.iter().map(|value| value.index).collect(),
Expand Down
3 changes: 3 additions & 0 deletions crates/spatialrust-py/tests/test_bindings.py
Original file line number Diff line number Diff line change
Expand Up @@ -193,6 +193,9 @@ def test_detection_nms_and_soft_nms():
assert indices[0] == 0
assert len(indices) == len(updated) == 3
assert updated[-1] < 0.8
view_indices, view_updated = sr.soft_nms(boxes, score_storage[::2], method="linear")
assert view_indices == indices
np.testing.assert_array_equal(view_updated, updated)


def test_mask_components_contours_and_rle():
Expand Down
97 changes: 96 additions & 1 deletion crates/spatialrust-vision/benches/detection.rs
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput};
use spatialrust_vision::{batched_nms, nms, BoundingBox2, Detection};
use spatialrust_vision::{batched_nms, nms, soft_nms, BoundingBox2, Detection, SoftNmsMethod};
use std::cmp::Ordering;

fn benchmark_nms(c: &mut Criterion) {
let mut group = c.benchmark_group("nms_xyxy_f32");
Expand Down Expand Up @@ -38,6 +39,100 @@ fn benchmark_nms(c: &mut Criterion) {
});
}
group.finish();

let mut group = c.benchmark_group("soft_nms_linear_xyxy_f32");
group.sample_size(10);
for &count in &[100_usize, 1_000, 8_400] {
let (boxes, scores) = detections(count);
group.throughput(Throughput::Elements(count as u64));
group.bench_function(BenchmarkId::from_parameter(count), |b| {
b.iter(|| {
black_box(
soft_nms(
black_box(&boxes),
black_box(&scores),
black_box(0.25),
black_box(0.5),
black_box(SoftNmsMethod::Linear),
)
.unwrap(),
)
});
});
group.bench_function(BenchmarkId::new("sorting_baseline", count), |b| {
b.iter(|| {
black_box(soft_nms_sorting_baseline(
black_box(&boxes),
black_box(&scores),
black_box(0.25),
black_box(0.5),
))
});
});
}
group.finish();
}

#[derive(Clone, Copy)]
struct BaselineCandidate {
index: usize,
score: f32,
}

fn soft_nms_sorting_baseline(
boxes: &[BoundingBox2],
scores: &[f32],
score_threshold: f32,
iou_threshold: f32,
) -> Vec<BaselineCandidate> {
let mut candidates = scores
.iter()
.copied()
.enumerate()
.map(|(index, score)| BaselineCandidate { index, score })
.collect::<Vec<_>>();
let areas = boxes.iter().copied().map(BoundingBox2::area).collect::<Vec<_>>();
let mut output = Vec::with_capacity(candidates.len());
while !candidates.is_empty() {
candidates.sort_by(|left, right| {
right
.score
.partial_cmp(&left.score)
.unwrap_or(Ordering::Equal)
.then_with(|| left.index.cmp(&right.index))
});
let selected = candidates.remove(0);
if selected.score < score_threshold {
break;
}
output.push(selected);
for candidate in &mut candidates {
let overlap = cached_iou(
boxes[selected.index],
areas[selected.index],
boxes[candidate.index],
areas[candidate.index],
);
if overlap > iou_threshold {
candidate.score *= 1.0 - overlap;
}
}
candidates.retain(|candidate| candidate.score >= score_threshold);
}
output
}

fn cached_iou(left: BoundingBox2, left_area: f32, right: BoundingBox2, right_area: f32) -> f32 {
let width = left.x_max.min(right.x_max) - left.x_min.max(right.x_min);
if width <= 0.0 {
return 0.0;
}
let height = left.y_max.min(right.y_max) - left.y_min.max(right.y_min);
if height <= 0.0 {
return 0.0;
}
let intersection = width * height;
intersection / (left_area + right_area - intersection)
}

fn detections(count: usize) -> (Vec<BoundingBox2>, Vec<f32>) {
Expand Down
Loading
Loading