diff --git a/README.md b/README.md index 80c9e4c..98aac81 100644 --- a/README.md +++ b/README.md @@ -192,6 +192,19 @@ These Windows-host medians include each Python API call and returned indices; see the [NMS harness](bench/opencv_nms_comparison/) and dated [receipt](notes/2026-07-15_nms_opencv_acceleration.md). +Class-aware post-processing uses the same exact-index gate against OpenCV +`dnn.NMSBoxesBatched`. SpatialRust stores kept indices by class, so candidates +never scan already-kept boxes from unrelated classes: + +| Batched NMS profile | OpenCV | SpatialRust | Result | +| --- | ---: | ---: | ---: | +| 1,000 candidates / 20 classes | 3.538 ms | 0.134 ms | **SpatialRust 26.38×** | +| 8,400 candidates / 80 classes | 211.762 ms | 2.178 ms | **SpatialRust 97.25×** | + +Both profiles returned exactly the same globally score-ordered indices. See +the [batched NMS harness](bench/opencv_batched_nms_comparison/) and dated +[receipt](notes/2026-07-15_batched_nms_opencv_acceleration.md). + #### Vision accuracy The same deterministic RGB inputs passed all VGA, 1080p, and 4K gates: diff --git a/bench/opencv_batched_nms_comparison/README.md b/bench/opencv_batched_nms_comparison/README.md new file mode 100644 index 0000000..951a603 --- /dev/null +++ b/bench/opencv_batched_nms_comparison/README.md @@ -0,0 +1,16 @@ +# OpenCV class-aware batched NMS comparison + +This harness compares SpatialRust `batched_nms` with OpenCV +`dnn.NMSBoxesBatched` using the same deterministic float32 boxes, scores, +integer class IDs, score threshold, and IoU threshold. It covers 1,000 +candidates across 20 classes and a YOLO-style 8,400 candidates across 80 +classes. Returned indices must match exactly before timings are published. + +```powershell +python bench/opencv_batched_nms_comparison/performance.py ` + --output target/opencv-batched-nms-performance.json +``` + +The report follows `spatialrust.opencv-comparison.v1` and records raw samples, +dispersion, library versions, thread policy, and the host environment. Results +are machine-specific and must not be generalized beyond the named workload. diff --git a/bench/opencv_batched_nms_comparison/performance.py b/bench/opencv_batched_nms_comparison/performance.py new file mode 100644 index 0000000..8945431 --- /dev/null +++ b/bench/opencv_batched_nms_comparison/performance.py @@ -0,0 +1,137 @@ +"""Reproducible class-aware batched NMS comparison with OpenCV.""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +import cv2 +import numpy as np +import spatialrust as sr + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from opencv_comparison.report import emit_report, environment, make_report, timed_pair + + +PROFILES = { + "multi_class_1000": (1_000, 20, 30), + "yolo_8400": (8_400, 80, 10), +} + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("--output", type=Path) + parser.add_argument("--profiles", default=",".join(PROFILES)) + parser.add_argument("--warmup", type=int, default=3) + return parser.parse_args() + + +def main() -> None: + args = parse_args() + selected = [name.strip() for name in args.profiles.split(",") if name.strip()] + unknown = sorted(set(selected) - PROFILES.keys()) + if unknown: + raise ValueError(f"unknown profiles: {', '.join(unknown)}") + if args.warmup < 0: + raise ValueError("warmup must be non-negative") + if not hasattr(cv2.dnn, "NMSBoxesBatched"): + raise RuntimeError("OpenCV build does not expose dnn.NMSBoxesBatched") + if hasattr(cv2, "ocl"): + cv2.ocl.setUseOpenCL(False) + + rng = np.random.default_rng(119) + results: dict[str, object] = {} + for profile in selected: + count, class_count, repeats = PROFILES[profile] + centers = rng.uniform(0.0, 640.0, size=(count, 2)).astype(np.float32) + sizes = rng.uniform(5.0, 120.0, size=(count, 2)).astype(np.float32) + boxes_xyxy = np.empty((count, 4), dtype=np.float32) + boxes_xyxy[:, :2] = centers - sizes * 0.5 + boxes_xyxy[:, 2:] = centers + sizes * 0.5 + boxes_xywh = boxes_xyxy.copy() + boxes_xywh[:, 2:] -= boxes_xywh[:, :2] + scores = rng.random(count, dtype=np.float32) + class_ids_cv = rng.integers(0, class_count, count, dtype=np.int32) + class_ids_sr = class_ids_cv.astype(np.int64) + + def opencv_batched_nms() -> np.ndarray: + return np.asarray( + cv2.dnn.NMSBoxesBatched( + boxes_xywh, class_ids=class_ids_cv, scores=scores, + score_threshold=0.25, nms_threshold=0.5, + ) + ).reshape(-1) + + def spatialrust_batched_nms() -> np.ndarray: + return sr.batched_nms(boxes_xyxy, scores, class_ids_sr, 0.25, 0.5) + + expected = opencv_batched_nms().astype(np.int64, copy=False) + actual = spatialrust_batched_nms() + exact = bool(np.array_equal(expected, actual)) + if not exact: + raise AssertionError(f"{profile} batched NMS index mismatch") + + _, _, opencv_timing, spatialrust_timing = timed_pair( + opencv_batched_nms, + spatialrust_batched_nms, + warmup=args.warmup, + repeats=repeats, + seed=121, + min_sample_time_ms=1.0, + ) + opencv_ms = float(opencv_timing["median"]) + spatialrust_ms = float(spatialrust_timing["median"]) + results[profile] = { + "box_count": count, + "class_count": class_count, + "kept_count": int(actual.size), + "score_threshold": 0.25, + "iou_threshold": 0.5, + "indices_exact": exact, + "opencv": opencv_timing, + "spatialrust": spatialrust_timing, + "spatialrust_speedup": opencv_ms / spatialrust_ms, + "faster_implementation": ( + "spatialrust" if spatialrust_ms < opencv_ms else "opencv" + ), + } + + receipt = environment( + opencv_version=cv2.__version__, spatialrust_version=sr.__version__ + ) + receipt["opencv_threads"] = cv2.getNumThreads() + receipt["opencv_opencl_enabled"] = bool( + hasattr(cv2, "ocl") and cv2.ocl.useOpenCL() + ) + report = make_report( + suite="opencv-batched-nms-performance", + kind="performance", + status="pass", + environment_receipt=receipt, + results={ + "methodology": { + "timing_scope": "Python API call returning globally score-ordered kept indices", + "paired_interleaved": True, + "input_seed": 119, + "random_order_seed": 121, + "minimum_sample_time_ms": 1.0, + "box_format": { + "opencv": "xywh float32 NumPy array", + "spatialrust": "xyxy float32 NumPy array", + }, + "class_id_format": { + "opencv": "int32 NumPy array", + "spatialrust": "int64 NumPy array", + }, + "thread_policy": "library defaults; OpenCV thread count recorded", + }, + "profiles": results, + }, + ) + emit_report(report, args.output) + + +if __name__ == "__main__": + main() diff --git a/bench/opencv_comparison/README.md b/bench/opencv_comparison/README.md index 69c4e7a..d188b9e 100644 --- a/bench/opencv_comparison/README.md +++ b/bench/opencv_comparison/README.md @@ -32,7 +32,7 @@ VGA, 1080p, and 4K profiles and the initial competitive workload set: 10. colored RGB-D to point cloud 11. AI preprocessing 12. RGB-D to voxel end-to-end -13. detection NMS post-processing +13. detection NMS and class-aware batched NMS post-processing Exact matches use a JSON `null` PSNR (mathematically infinite) so reports remain strict RFC-compatible JSON. Numerical comparisons retain max/mean/RMS and @@ -51,6 +51,7 @@ then run both current suites: ```powershell python bench\opencv_comparison\run.py python bench\opencv_nms_comparison\performance.py +python bench\opencv_batched_nms_comparison\performance.py ``` Reports are written under `target/opencv-comparison/`. Run one suite with diff --git a/bench/opencv_comparison/manifest.json b/bench/opencv_comparison/manifest.json index c15dec3..3f89177 100644 --- a/bench/opencv_comparison/manifest.json +++ b/bench/opencv_comparison/manifest.json @@ -48,6 +48,7 @@ { "id": "rgbd_to_point_cloud", "domain": "spatial-e2e", "modes": ["allocate"] }, { "id": "ai_preprocess", "domain": "dnn-adapter", "modes": ["allocate", "reuse"] }, { "id": "nms", "domain": "dnn-adapter", "modes": ["postprocess"] }, + { "id": "batched_nms", "domain": "dnn-adapter", "modes": ["postprocess"] }, { "id": "rgbd_to_voxel", "domain": "spatial-e2e", "modes": ["allocate"] } ] } diff --git a/bench/opencv_comparison/test_report.py b/bench/opencv_comparison/test_report.py index 69f0e94..f6c019c 100644 --- a/bench/opencv_comparison/test_report.py +++ b/bench/opencv_comparison/test_report.py @@ -118,6 +118,7 @@ def test_manifest_reserves_representative_profiles_and_workloads(self) -> None: self.assertIn("rgbd_to_voxel", workloads) self.assertIn("ai_preprocess", workloads) self.assertIn("nms", workloads) + self.assertIn("batched_nms", workloads) self.assertIn("coefficient_of_variation", statistics) self.assertIn("median_absolute_deviation", statistics) self.assertIn("batch_size", statistics) diff --git a/crates/spatialrust-py/README.md b/crates/spatialrust-py/README.md index 91777f9..734cdc8 100644 --- a/crates/spatialrust-py/README.md +++ b/crates/spatialrust-py/README.md @@ -107,7 +107,7 @@ reloaded = sr.read("labeled.las") | `rgbd_to_point_cloud(depth, color, fx, fy, cx, cy, ...)` | Aligned `(H,W)` depth + `(H,W,3)` RGB to an XYZRGB cloud | | `resize_image` / `letterbox_image` / `normalize_image_chw` | Model-ready RGB resize, padding, and float32 CHW packing | | `rgb_to_gray_image` / `rgb_to_hsv_image` / `remap_image` | CPU color conversion and coordinate-map resampling | -| `nms` / `soft_nms` | Detection post-processing for `(N,4)` xyxy boxes | +| `nms` / `batched_nms` / `soft_nms` | Detection post-processing for `(N,4)` xyxy boxes, including class-aware suppression | | `connected_components_image` / `find_mask_contours` | Binary-mask labeling and contour extraction | | `encode_mask_rle` / `decode_mask_rle` | Row-major or COCO column-major binary-mask RLE | | `point_map_to_point_cloud` | Filter a dense `(H,W,3)` point map into a native point cloud | diff --git a/crates/spatialrust-py/spatialrust.pyi b/crates/spatialrust-py/spatialrust.pyi index a154bf0..766c10e 100644 --- a/crates/spatialrust-py/spatialrust.pyi +++ b/crates/spatialrust-py/spatialrust.pyi @@ -36,7 +36,7 @@ __all__: list[str] = [ "histogram_image", "equalize_histogram_image", "clahe_image", "integral_image_u8", "canny_image", "resize_image", "letterbox_image", "normalize_image_chw", "rgb_to_gray_image", "rgb_to_hsv_image", "remap_image", - "nms", "soft_nms", "connected_components_image", "distance_transform_edt", + "nms", "batched_nms", "soft_nms", "connected_components_image", "distance_transform_edt", "find_mask_contours", "encode_mask_rle", "decode_mask_rle", "point_map_to_point_cloud", "knn_graph", "radius_graph", "register_icp", "register_point_to_plane", "register_gicp", @@ -273,6 +273,13 @@ def nms( score_threshold: float = ..., iou_threshold: float = ..., ) -> NDArray[np.int64]: ... +def batched_nms( + boxes: _F32Array, + scores: _F32Array, + class_ids: NDArray[np.int64], + score_threshold: float = ..., + iou_threshold: float = ..., +) -> NDArray[np.int64]: ... def soft_nms( boxes: _F32Array, scores: _F32Array, diff --git a/crates/spatialrust-py/src/lib.rs b/crates/spatialrust-py/src/lib.rs index 3a31cdf..5c3b180 100644 --- a/crates/spatialrust-py/src/lib.rs +++ b/crates/spatialrust-py/src/lib.rs @@ -71,8 +71,8 @@ use spatialrust::transform::{ }; use spatialrust::vision::{ adaptive_threshold as adaptive_threshold_op, approximate_polygon as approximate_contour, - bilateral_filter as bilateral_filter_op, canny as canny_op, clahe as clahe_op, - connected_components as label_components, decode_rle as decode_mask_runs, + batched_nms as batched_nms_op, bilateral_filter as bilateral_filter_op, canny as canny_op, + clahe as clahe_op, connected_components as label_components, decode_rle as decode_mask_runs, detect_and_describe_orb as detect_and_describe_orb_op, detect_fast as detect_fast_op, detect_harris as detect_harris_op, detect_shi_tomasi as detect_shi_tomasi_op, distance_transform_edt_u8_into as distance_transform_edt_u8_into_op, @@ -93,7 +93,7 @@ use spatialrust::vision::{ solve_pnp as solve_pnp_op, stereo_block_match as stereo_block_match_op, stitch_panorama_pair as stitch_panorama_pair_op, threshold as threshold_op, AbsolutePose, AdaptiveThresholdMethod, BinaryMask, BorderMode, BoundingBox2, CameraMatrix3, CannyOptions, - ConfidenceMap, Connectivity, CornerSelectionOptions, DescriptorBuffer, + ConfidenceMap, Connectivity, CornerSelectionOptions, DescriptorBuffer, Detection, DistanceTransformWorkspace, FastOptions, HarrisOptions, Interpolation, Kernel2D, Keypoint2, MaskRle, MatchOptions, MorphologyOperation, MorphologyShape, ObjectImageCorrespondence, OrbOptions, OrbScoreType, PanoramaOptions, PerspectiveTransform, PointCorrespondence2, @@ -3055,6 +3055,59 @@ fn nms<'py>( Ok(indices.into_pyarray_bound(py)) } +/// Class-aware greedy NMS over `(N, 4)` xyxy boxes. +#[pyfunction] +#[pyo3(signature = (boxes, scores, class_ids, score_threshold=0.0, iou_threshold=0.5))] +fn batched_nms<'py>( + py: Python<'py>, + boxes: PyReadonlyArray2<'_, f32>, + scores: PyReadonlyArray1<'_, f32>, + class_ids: PyReadonlyArray1<'_, i64>, + score_threshold: f32, + iou_threshold: f32, +) -> PyResult>> { + let boxes_view = boxes.as_array(); + if boxes_view.shape().len() != 2 || boxes_view.shape()[1] != 4 { + return Err(PyValueError::new_err("expected boxes with shape (N, 4)")); + } + let scores_view = scores.as_array(); + let packed_scores; + let scores = match scores_view.as_slice() { + Some(scores) => scores, + None => { + packed_scores = scores_view.iter().copied().collect::>(); + packed_scores.as_slice() + } + }; + let class_ids_view = class_ids.as_array(); + let packed_class_ids; + let class_ids = match class_ids_view.as_slice() { + Some(class_ids) => class_ids, + None => { + packed_class_ids = class_ids_view.iter().copied().collect::>(); + packed_class_ids.as_slice() + } + }; + let count = boxes_view.shape()[0]; + if scores.len() != count || class_ids.len() != count { + return Err(PyValueError::new_err("boxes, scores, and class_ids must have equal lengths")); + } + let mut detections = Vec::with_capacity(count); + for (index, row) in boxes_view.rows().into_iter().enumerate() { + detections.push(Detection { + bbox: BoundingBox2::try_new(row[0], row[1], row[2], row[3]).map_err(to_py_err)?, + score: scores[index], + class_id: class_ids[index], + }); + } + let indices = batched_nms_op(&detections, score_threshold, iou_threshold) + .map_err(to_py_err)? + .into_iter() + .map(|index| index as i64) + .collect::>(); + Ok(indices.into_pyarray_bound(py)) +} + /// Soft-NMS returning `(indices, updated_scores)`. #[pyfunction] #[pyo3(signature = (boxes, scores, score_threshold=0.001, iou_threshold=0.5, method="linear", sigma=0.5))] @@ -3424,6 +3477,7 @@ fn spatialrust_module(m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_function(wrap_pyfunction!(rgb_to_hsv_image, m)?)?; m.add_function(wrap_pyfunction!(remap_image, m)?)?; m.add_function(wrap_pyfunction!(nms, m)?)?; + m.add_function(wrap_pyfunction!(batched_nms, m)?)?; m.add_function(wrap_pyfunction!(soft_nms, m)?)?; m.add_function(wrap_pyfunction!(connected_components_image, m)?)?; m.add_function(wrap_pyfunction!(distance_transform_edt, m)?)?; diff --git a/crates/spatialrust-py/tests/test_bindings.py b/crates/spatialrust-py/tests/test_bindings.py index 4283eac..759a300 100644 --- a/crates/spatialrust-py/tests/test_bindings.py +++ b/crates/spatialrust-py/tests/test_bindings.py @@ -66,7 +66,7 @@ def test_exports_present(): "rgbd_to_point_cloud", "depth_to_xyz", "resize_image", "letterbox_image", "normalize_image_chw", "rgb_to_gray_image", "rgb_to_hsv_image", "remap_image", - "nms", "soft_nms", "connected_components_image", "distance_transform_edt", + "nms", "batched_nms", "soft_nms", "connected_components_image", "distance_transform_edt", "find_mask_contours", "encode_mask_rle", "decode_mask_rle", "point_map_to_point_cloud", ): @@ -179,6 +179,16 @@ def test_detection_nms_and_soft_nms(): score_storage = np.empty(scores.size * 2, dtype=np.float32) score_storage[::2] = scores np.testing.assert_array_equal(sr.nms(boxes, score_storage[::2]), [0, 2]) + np.testing.assert_array_equal( + sr.batched_nms(boxes, scores, np.array([4, 4, 4], dtype=np.int64)), [0, 2] + ) + class_storage = np.zeros(scores.size * 2, dtype=np.int64) + class_storage[::2] = [4, 9, 4] + np.testing.assert_array_equal( + sr.batched_nms(boxes, score_storage[::2], class_storage[::2]), [0, 1, 2] + ) + with pytest.raises(ValueError, match="equal lengths"): + sr.batched_nms(boxes, scores[:-1], np.array([4, 9, 4], dtype=np.int64)) indices, updated = sr.soft_nms(boxes, scores, method="linear") assert indices[0] == 0 assert len(indices) == len(updated) == 3 diff --git a/crates/spatialrust-vision/benches/detection.rs b/crates/spatialrust-vision/benches/detection.rs index 154bee1..3cbd82e 100644 --- a/crates/spatialrust-vision/benches/detection.rs +++ b/crates/spatialrust-vision/benches/detection.rs @@ -1,5 +1,5 @@ use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; -use spatialrust_vision::{nms, BoundingBox2}; +use spatialrust_vision::{batched_nms, nms, BoundingBox2, Detection}; fn benchmark_nms(c: &mut Criterion) { let mut group = c.benchmark_group("nms_xyxy_f32"); @@ -17,6 +17,27 @@ fn benchmark_nms(c: &mut Criterion) { }); } group.finish(); + + let mut group = c.benchmark_group("batched_nms_xyxy_f32_80_classes"); + group.sample_size(10); + for &count in &[1_000_usize, 8_400] { + let (boxes, scores) = detections(count); + let detections = boxes + .into_iter() + .zip(scores) + .enumerate() + .map(|(index, (bbox, score))| Detection { bbox, score, class_id: (index % 80) as i64 }) + .collect::>(); + group.throughput(Throughput::Elements(count as u64)); + group.bench_function(BenchmarkId::from_parameter(count), |b| { + b.iter(|| { + black_box( + batched_nms(black_box(&detections), black_box(0.25), black_box(0.5)).unwrap(), + ) + }); + }); + } + group.finish(); } fn detections(count: usize) -> (Vec, Vec) { diff --git a/crates/spatialrust-vision/src/detection.rs b/crates/spatialrust-vision/src/detection.rs index 7beaa22..efc9e31 100644 --- a/crates/spatialrust-vision/src/detection.rs +++ b/crates/spatialrust-vision/src/detection.rs @@ -1,6 +1,7 @@ //! Detection post-processing primitives. use std::cmp::Ordering; +use std::collections::HashMap; use crate::{VisionError, VisionResult}; @@ -216,20 +217,23 @@ pub fn batched_nms( sort_indices_by_score(&mut order, &scores); let areas: Vec = detections.iter().map(|detection| detection.bbox.area()).collect(); let mut keep: Vec = Vec::with_capacity(order.len()); + let mut keep_by_class: HashMap> = HashMap::new(); 'candidate: for index in order { - for &selected in &keep { - if detections[index].class_id == detections[selected].class_id - && iou_with_areas( - detections[index].bbox, - areas[index], - detections[selected].bbox, - areas[selected], - ) > iou_threshold + let class_id = detections[index].class_id; + let selected_for_class = keep_by_class.entry(class_id).or_default(); + for &selected in selected_for_class.iter() { + if iou_with_areas( + detections[index].bbox, + areas[index], + detections[selected].bbox, + areas[selected], + ) > iou_threshold { continue 'candidate; } } keep.push(index); + selected_for_class.push(index); } Ok(keep) } @@ -398,6 +402,68 @@ mod tests { assert_eq!(batched_nms(&detections, 0.0, 0.5).unwrap(), vec![0, 1]); } + #[test] + fn batched_nms_suppresses_per_class_in_global_score_order() { + let detections = [ + Detection { bbox: bbox(0.0, 0.0, 2.0, 2.0), score: 0.7, class_id: 4 }, + Detection { bbox: bbox(0.1, 0.1, 2.1, 2.1), score: 0.9, class_id: 4 }, + Detection { bbox: bbox(0.1, 0.1, 2.1, 2.1), score: 0.8, class_id: 9 }, + Detection { bbox: bbox(5.0, 5.0, 6.0, 6.0), score: 0.6, class_id: 4 }, + ]; + assert_eq!(batched_nms(&detections, 0.0, 0.5).unwrap(), vec![1, 2, 3]); + } + + #[test] + fn class_buckets_match_global_scan_reference() { + let mut state = 119_u64; + let detections = (0..257) + .map(|index| { + let x = sample(&mut state) * 64.0; + let y = sample(&mut state) * 64.0; + let width = 1.0 + sample(&mut state) * 20.0; + let height = 1.0 + sample(&mut state) * 20.0; + Detection { + bbox: bbox(x, y, x + width, y + height), + score: sample(&mut state), + class_id: (index % 11) as i64 - 5, + } + }) + .collect::>(); + let actual = batched_nms(&detections, 0.2, 0.45).unwrap(); + + let mut order = (0..detections.len()) + .filter(|&index| detections[index].score >= 0.2) + .collect::>(); + order.sort_by(|&left, &right| { + detections[right] + .score + .partial_cmp(&detections[left].score) + .unwrap() + .then_with(|| left.cmp(&right)) + }); + let mut expected: Vec = Vec::new(); + 'candidate: for index in order { + for &selected in &expected { + if detections[index].class_id == detections[selected].class_id + && detections[index].bbox.iou(detections[selected].bbox) > 0.45 + { + continue 'candidate; + } + } + expected.push(index); + } + assert_eq!(actual, expected); + } + + fn sample(state: &mut u64) -> f32 { + *state = state.wrapping_add(0x9E37_79B9_7F4A_7C15); + let mut value = *state; + value = (value ^ (value >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + value = (value ^ (value >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + value ^= value >> 31; + (value >> 40) as f32 / (1_u32 << 24) as f32 + } + #[test] fn soft_nms_decays_overlapping_score() { let boxes = [bbox(0.0, 0.0, 2.0, 2.0), bbox(0.1, 0.1, 2.1, 2.1)]; diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 47d3c33..7226c3b 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -619,6 +619,7 @@ to one implicitly, and GPU receipts must retain named upload/readback stages. | 114E | Complete | Exact EDT binary-row fast path, tiled transpose, and bounded pool dispatch | VGA/1080p/4K Criterion and OpenCV receipt | | 114F | Complete | Cache EDT parabola heights and balance column tasks for dense masks | exact OpenCV parity and 4K Python reuse win | | 114G | Complete | Cache NMS box geometry and avoid packed Python score copies | exact OpenCV index parity and 100/1,000/8,400-candidate wins | +| 114H | Complete | Bucket class-aware NMS keeps and expose one-call Python batched NMS | exact OpenCV parity and 26.38×/97.25× wins | ### Epic 115 delivery slices diff --git a/docs/site/algorithms.html b/docs/site/algorithms.html index 176870d..babf0f0 100644 --- a/docs/site/algorithms.html +++ b/docs/site/algorithms.html @@ -36,7 +36,7 @@

Algorithm catalog

Image analysisFixed/Otsu/adaptive threshold, histogram, equalization, CLAHE, integral image, Cannyspatialrust-vision · imgproc-analysis, imgproc-cannyCPU Local featuresFAST, Harris, Shi–Tomasi, ORB, descriptor matching, grid selection, pyramidal Lucas–Kanade trackingspatialrust-vision · feature2dCPU Dense visionExact Euclidean distance transform with reusable workspace/output, connected components, contours, polygon approximation, mask RLE, depth/confidence/flow/point mapsspatialrust-vision · denseCPU parallel - Detection post-processingIoU/GIoU, greedy NMS, class-aware batched NMS, and hard/linear/Gaussian Soft-NMS. Seeded Python NMS is 3.22×–8.95× faster than OpenCV on the recorded 100/1,000/8,400-candidate workloads with exact kept indices. Harness and methodology.spatialrust-vision · detectionCPU + Detection post-processingIoU/GIoU, greedy NMS, class-aware batched NMS, and hard/linear/Gaussian Soft-NMS. Seeded Python NMS is 3.22×–8.95× faster than OpenCV; class-aware batched NMS is 26.38×–97.25× faster on the recorded 1,000/8,400-candidate workloads. Both require exact kept-index parity. NMS harness · batched harness.spatialrust-vision · detectionCPU Multiview geometryHomography, fundamental/essential matrices, RANSAC, triangulation, relative pose, PnP/PnP-RANSACspatialrust-vision · geometryCPU Stereo and odometryStereo rectification, block matching, disparity-to-depth/XYZ, monocular and RGB-D visual odometryspatialrust-vision · geometry, odometryCPU VideoDense block flow, adaptive background model, multi-object tracker, timestamped pull sourcesspatialrust-vision · videoCPU diff --git a/docs/site/vision2.html b/docs/site/vision2.html index 89da5ac..10e081e 100644 --- a/docs/site/vision2.html +++ b/docs/site/vision2.html @@ -51,7 +51,7 @@

Latest measured outcome

Exact EDT at 4K

With caller-owned output and workspace reuse, SpatialRust measured 40.66 ms versus OpenCV 43.33 ms: a 1.07× lead on the recorded Windows host.

Accuracy preserved

The VGA, 1080p, and 4K canonical masks retain maximum absolute error 0.0 against OpenCV's precise L2 distance transform.

-

NMS post-processing

SpatialRust measured 3.22× faster for 8,400 YOLO-style candidates and up to 8.95× faster for 100 candidates, with kept indices exactly matching OpenCV.

+

NMS post-processing

SpatialRust measured 3.22×–8.95× faster for class-agnostic NMS and 26.38×–97.25× faster for class-aware batched NMS, with kept indices exactly matching OpenCV on every recorded profile.

This is a workload- and host-specific result. VGA and 1080p reuse remain narrow OpenCV wins; the repository receipt contains the reproducible methodology.

diff --git a/notes/2026-07-15_batched_nms_opencv_acceleration.md b/notes/2026-07-15_batched_nms_opencv_acceleration.md new file mode 100644 index 0000000..91e9053 --- /dev/null +++ b/notes/2026-07-15_batched_nms_opencv_acceleration.md @@ -0,0 +1,56 @@ +# Class-aware batched NMS OpenCV acceleration receipt — 2026-07-15 + +## Outcome + +SpatialRust class-aware NMS now stores accepted indices in per-class buckets. +Each candidate only performs IoU comparisons with previously accepted boxes +from its own class, while a separate output vector preserves global descending +score order. The public Rust API remains safe and deterministic. + +Python now exposes `batched_nms(boxes, scores, class_ids, ...)`. Contiguous +float32 scores and int64 class IDs are borrowed; non-contiguous arrays use an +explicit packed fallback. The call returns original indices as int64 NumPy. + +## Correctness contract + +The comparison uses OpenCV +[`dnn.NMSBoxesBatched`](https://docs.opencv.org/master/df/d57/namespacecv_1_1dnn.html) +as the oracle. Both implementations receive the same deterministic float32 +boxes and scores, integer class IDs, score threshold `0.25`, and IoU threshold +`0.5`. OpenCV receives `xywh`; SpatialRust receives the corresponding `xyxy`. + +| Profile | Candidates | Classes | Kept | Exact ordered indices | +| --- | ---: | ---: | ---: | ---: | +| Multi-class | 1,000 | 20 | 733 | yes | +| YOLO-style | 8,400 | 80 | 6,234 | yes | + +Rust coverage also compares the bucketed implementation with the previous +global-scan semantics on 257 deterministic boxes across positive and negative +class IDs. Python coverage checks same-class suppression, different-class +retention, non-contiguous scores/class IDs, and length validation. + +## Performance + +Host: Windows 11, Intel 6-core/12-thread CPU, CPython 3.12.10, OpenCV 4.10, +OpenCL disabled. Timings are seeded, randomized, interleaved Python API medians +after three warmups and include the returned index array. + +| Candidates / classes | Repeats | OpenCV | SpatialRust | SpatialRust speedup | +| ---: | ---: | ---: | ---: | ---: | +| 1,000 / 20 | 30 | 3.5377 ms | 0.1341 ms | **26.38×** | +| 8,400 / 80 | 10 | 211.7618 ms | 2.1776 ms | **97.25×** | + +The machine-readable report retains raw samples and dispersion; the 8,400 +profile had visible host-load variation, but every paired sample set retained +a wide SpatialRust lead. These are scoped workload results, not universal +claims about all inputs or OpenCV builds. + +Native Criterion medians for a separately seeded 80-class workload were about +98.3 microseconds for 1,000 candidates and 2.42 milliseconds for 8,400. + +Reproduce the report with: + +```powershell +python bench/opencv_batched_nms_comparison/performance.py ` + --output target/opencv-batched-nms-performance.json +```