From c6ecbf05ee0e053a860ab568557e7218316cb6de Mon Sep 17 00:00:00 2001 From: rsasaki0109 Date: Wed, 15 Jul 2026 16:12:28 +0900 Subject: [PATCH] Add Epic 103 reusable CPU vision kernels --- CHANGELOG.md | 7 + bench/opencv_comparison/manifest.json | 6 +- bench/opencv_comparison/run.py | 4 + bench/opencv_vision_comparison/README.md | 9 + bench/opencv_vision_comparison/performance.py | 219 ++++++++++++++++++ crates/spatialrust-image/src/lib.rs | 7 + crates/spatialrust-platform/src/stability.rs | 6 +- crates/spatialrust-py/spatialrust.pyi | 4 +- crates/spatialrust-py/src/lib.rs | 187 ++++++++++----- crates/spatialrust-py/tests/test_bindings.py | 9 + crates/spatialrust-vision/Cargo.toml | 5 + .../spatialrust-vision/benches/preprocess.rs | 42 +++- crates/spatialrust-vision/benches/resize.rs | 41 ++++ crates/spatialrust-vision/src/preprocess.rs | 218 ++++++++++++++--- crates/spatialrust-vision/src/resize.rs | 104 +++++++-- crates/spatialrust/tests/vision_api_v1.rs | 24 +- docs/API_STABILITY.md | 2 +- docs/ROADMAP.md | 18 +- notes/2026-07-15_epic103_cpu_vision.md | 20 ++ 19 files changed, 818 insertions(+), 114 deletions(-) create mode 100644 bench/opencv_vision_comparison/performance.py create mode 100644 crates/spatialrust-vision/benches/resize.rs create mode 100644 notes/2026-07-15_epic103_cpu_vision.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 568d3cb..f2f70dc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,6 +21,13 @@ removed no sooner than the next major (see `docs/API_STABILITY.md`). ### Added +- **Reusable CPU vision kernels (Epic 103)**: stable caller-owned + `resize_into`, `rgb_to_gray_into`, `normalize_into`, and `pack_chw_into` + paths, NumPy `out=` bindings, size-aware safe parallel CHW packing, and a + VGA/1080p/4K allocation/reuse comparison receipt. The reference run records + both OpenCV's resize/color-conversion lead and SpatialRust's 8.54x–16.11x + reusable RGB-to-CHW advantage. + - **Vision API conformance (Epic 102)**: a machine-readable stable/provisional image-camera-vision registry, a compile-and-behavior API contract, and a dedicated Linux/Windows/macOS CI matrix for the image, camera, and full vision diff --git a/bench/opencv_comparison/manifest.json b/bench/opencv_comparison/manifest.json index 9786fd1..fc69d69 100644 --- a/bench/opencv_comparison/manifest.json +++ b/bench/opencv_comparison/manifest.json @@ -15,8 +15,8 @@ "spatialrust_version" ], "workloads": [ - { "id": "resize_bilinear", "domain": "imgproc", "modes": ["allocate"] }, - { "id": "rgb_to_gray", "domain": "imgproc", "modes": ["allocate"] }, + { "id": "resize_bilinear", "domain": "imgproc", "modes": ["allocate", "reuse"] }, + { "id": "rgb_to_gray", "domain": "imgproc", "modes": ["allocate", "reuse"] }, { "id": "gaussian_blur", "domain": "imgproc", "modes": ["allocate"] }, { "id": "sobel", "domain": "imgproc", "modes": ["allocate"] }, { "id": "canny", "domain": "imgproc", "modes": ["allocate"] }, @@ -25,7 +25,7 @@ { "id": "stereo_bm", "domain": "calib3d", "modes": ["allocate"] }, { "id": "depth_to_xyz", "domain": "rgbd", "modes": ["allocate", "reuse"] }, { "id": "rgbd_to_point_cloud", "domain": "spatial-e2e", "modes": ["allocate"] }, - { "id": "ai_preprocess", "domain": "dnn-adapter", "modes": ["allocate"] }, + { "id": "ai_preprocess", "domain": "dnn-adapter", "modes": ["allocate", "reuse"] }, { "id": "rgbd_to_voxel", "domain": "spatial-e2e", "modes": ["allocate"] } ] } diff --git a/bench/opencv_comparison/run.py b/bench/opencv_comparison/run.py index f258ecd..401a0ed 100644 --- a/bench/opencv_comparison/run.py +++ b/bench/opencv_comparison/run.py @@ -13,6 +13,10 @@ ROOT = Path(__file__).resolve().parents[2] SUITES = { "vision": ROOT / "bench" / "opencv_vision_comparison" / "run.py", + "vision-performance": ROOT + / "bench" + / "opencv_vision_comparison" + / "performance.py", "rgbd": ROOT / "bench" / "opencv_rgbd_comparison" / "run.py", } diff --git a/bench/opencv_vision_comparison/README.md b/bench/opencv_vision_comparison/README.md index 9f036dc..7fbf239 100644 --- a/bench/opencv_vision_comparison/README.md +++ b/bench/opencv_vision_comparison/README.md @@ -22,3 +22,12 @@ when a documented numerical tolerance is exceeded. Pass `--output PATH` to retain the report. OpenCV is comparison/test tooling only; it is not a Rust runtime dependency. The shared report contract and workload registry are in [`../opencv_comparison`](../opencv_comparison/README.md). + +Epic 103 adds allocate/reuse timing for bilinear resize, RGB-to-gray, and AI +CHW preprocessing at VGA, 1080p, and 4K. It preserves raw samples and p95 in +the same report contract: + +```powershell +python bench\opencv_vision_comparison\performance.py ` + --output target\opencv-comparison\vision-performance.json +``` diff --git a/bench/opencv_vision_comparison/performance.py b/bench/opencv_vision_comparison/performance.py new file mode 100644 index 0000000..5a2e591 --- /dev/null +++ b/bench/opencv_vision_comparison/performance.py @@ -0,0 +1,219 @@ +"""Performance comparison for Epic 103 reusable CPU vision paths.""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +import cv2 +import numpy as np +import spatialrust as sr + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from opencv_comparison.report import emit_report, environment, make_report, timed + + +PROFILES = { + "vga": (640, 480, 20), + "1080p": (1920, 1080, 8), + "4k": (3840, 2160, 3), +} + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("--output", type=Path) + parser.add_argument("--profiles", default="vga,1080p,4k") + parser.add_argument("--warmup", type=int, default=3) + return parser.parse_args() + + +def measurement( + workload: str, + implementation: str, + mode: str, + width: int, + height: int, + timing: dict[str, object], +) -> dict[str, object]: + return { + "workload": workload, + "implementation": implementation, + "mode": mode, + "width": width, + "height": height, + "timing": timing, + } + + +def main() -> None: + args = parse_args() + selected = [name.strip() for name in args.profiles.split(",") if name.strip()] + unknown = sorted(set(selected) - PROFILES.keys()) + if unknown: + raise ValueError(f"unknown profiles: {', '.join(unknown)}") + if args.warmup < 0: + raise ValueError("warmup must be non-negative") + + if hasattr(cv2, "ocl"): + cv2.ocl.setUseOpenCL(False) + + rng = np.random.default_rng(103) + measurements: list[dict[str, object]] = [] + correctness: dict[str, object] = {} + speedups: dict[str, object] = {} + + for profile in selected: + width, height, repeats = PROFILES[profile] + image = rng.integers(0, 256, size=(height, width, 3), dtype=np.uint8) + output_width, output_height = width // 2, height // 2 + + resize_cv_out = np.empty((output_height, output_width, 3), dtype=np.uint8) + resize_sr_out = np.empty_like(resize_cv_out) + resize_cv = cv2.resize(image, (output_width, output_height), interpolation=cv2.INTER_LINEAR) + resize_sr = sr.resize_image(image, output_width, output_height, interpolation="bilinear") + resize_error = int( + np.max(np.abs(resize_cv.astype(np.int16) - resize_sr.astype(np.int16))) + ) + correctness[f"{profile}_resize_max_u8_error"] = resize_error + if resize_error > 1: + raise AssertionError(f"{profile} resize error {resize_error} > 1") + + _, cv_resize_alloc = timed( + lambda: cv2.resize(image, (output_width, output_height), interpolation=cv2.INTER_LINEAR), + warmup=args.warmup, + repeats=repeats, + ) + _, sr_resize_alloc = timed( + lambda: sr.resize_image(image, output_width, output_height, interpolation="bilinear"), + warmup=args.warmup, + repeats=repeats, + ) + _, cv_resize_reuse = timed( + lambda: cv2.resize( + image, + (output_width, output_height), + dst=resize_cv_out, + interpolation=cv2.INTER_LINEAR, + ), + warmup=args.warmup, + repeats=repeats, + ) + _, sr_resize_reuse = timed( + lambda: sr.resize_image( + image, + output_width, + output_height, + interpolation="bilinear", + out=resize_sr_out, + ), + warmup=args.warmup, + repeats=repeats, + ) + np.testing.assert_array_equal(resize_sr_out, resize_sr) + + gray_cv_out = np.empty((height, width), dtype=np.uint8) + gray_sr_out = np.empty_like(gray_cv_out) + gray_cv = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY) + gray_sr = sr.rgb_to_gray_image(image) + gray_error = int(np.max(np.abs(gray_cv.astype(np.int16) - gray_sr.astype(np.int16)))) + correctness[f"{profile}_rgb_to_gray_max_u8_error"] = gray_error + if gray_error > 1: + raise AssertionError(f"{profile} gray error {gray_error} > 1") + _, cv_gray_alloc = timed( + lambda: cv2.cvtColor(image, cv2.COLOR_RGB2GRAY), + warmup=args.warmup, + repeats=repeats, + ) + _, sr_gray_alloc = timed( + lambda: sr.rgb_to_gray_image(image), warmup=args.warmup, repeats=repeats + ) + _, cv_gray_reuse = timed( + lambda: cv2.cvtColor(image, cv2.COLOR_RGB2GRAY, dst=gray_cv_out), + warmup=args.warmup, + repeats=repeats, + ) + _, sr_gray_reuse = timed( + lambda: sr.rgb_to_gray_image(image, out=gray_sr_out), + warmup=args.warmup, + repeats=repeats, + ) + np.testing.assert_array_equal(gray_sr_out, gray_sr) + + chw_sr_out = np.empty((3, height, width), dtype=np.float32) + blob_cv = cv2.dnn.blobFromImage( + image, scalefactor=1.0 / 255.0, size=(width, height), swapRB=False, crop=False + )[0] + chw_sr = sr.normalize_image_chw(image) + chw_error = float(np.max(np.abs(blob_cv - chw_sr))) + correctness[f"{profile}_ai_preprocess_max_f32_error"] = chw_error + if chw_error > 1e-6: + raise AssertionError(f"{profile} AI preprocess error {chw_error} > 1e-6") + _, cv_chw_alloc = timed( + lambda: cv2.dnn.blobFromImage( + image, scalefactor=1.0 / 255.0, size=(width, height), swapRB=False, crop=False + ), + warmup=args.warmup, + repeats=repeats, + ) + _, sr_chw_alloc = timed( + lambda: sr.normalize_image_chw(image), warmup=args.warmup, repeats=repeats + ) + _, sr_chw_reuse = timed( + lambda: sr.normalize_image_chw(image, out=chw_sr_out), + warmup=args.warmup, + repeats=repeats, + ) + np.testing.assert_allclose(chw_sr_out, chw_sr, atol=0.0, rtol=0.0) + + rows = ( + ("resize_bilinear", "opencv", "allocate", cv_resize_alloc), + ("resize_bilinear", "spatialrust", "allocate", sr_resize_alloc), + ("resize_bilinear", "opencv", "reuse", cv_resize_reuse), + ("resize_bilinear", "spatialrust", "reuse", sr_resize_reuse), + ("rgb_to_gray", "opencv", "allocate", cv_gray_alloc), + ("rgb_to_gray", "spatialrust", "allocate", sr_gray_alloc), + ("rgb_to_gray", "opencv", "reuse", cv_gray_reuse), + ("rgb_to_gray", "spatialrust", "reuse", sr_gray_reuse), + ("ai_preprocess", "opencv", "allocate", cv_chw_alloc), + ("ai_preprocess", "spatialrust", "allocate", sr_chw_alloc), + ("ai_preprocess", "spatialrust", "reuse", sr_chw_reuse), + ) + measurements.extend( + measurement(workload, implementation, mode, width, height, timing) + for workload, implementation, mode, timing in rows + ) + speedups[profile] = { + "resize_allocate": cv_resize_alloc["median"] / sr_resize_alloc["median"], + "resize_reuse": cv_resize_reuse["median"] / sr_resize_reuse["median"], + "rgb_to_gray_allocate": cv_gray_alloc["median"] / sr_gray_alloc["median"], + "rgb_to_gray_reuse": cv_gray_reuse["median"] / sr_gray_reuse["median"], + "ai_preprocess_allocate": cv_chw_alloc["median"] / sr_chw_alloc["median"], + "ai_preprocess_reuse_vs_opencv_allocate": cv_chw_alloc["median"] + / sr_chw_reuse["median"], + } + + environment_receipt = environment( + opencv_version=cv2.__version__, spatialrust_version=sr.__version__ + ) + environment_receipt["opencv_threads"] = cv2.getNumThreads() + environment_receipt["opencv_opencl_enabled"] = bool( + hasattr(cv2, "ocl") and cv2.ocl.useOpenCL() + ) + report = make_report( + suite="opencv-vision-performance", + kind="performance", + status="pass", + environment_receipt=environment_receipt, + results={ + "correctness": correctness, + "speedup_vs_opencv": speedups, + "measurements": measurements, + }, + ) + emit_report(report, args.output) + + +if __name__ == "__main__": + main() diff --git a/crates/spatialrust-image/src/lib.rs b/crates/spatialrust-image/src/lib.rs index c39d07a..6e7b046 100644 --- a/crates/spatialrust-image/src/lib.rs +++ b/crates/spatialrust-image/src/lib.rs @@ -590,6 +590,13 @@ impl<'a, T, const CHANNELS: usize> ImageViewMut<'a, T, CHANNELS> { self.metadata } + /// Replaces semantic metadata after validating the channel count. + pub fn set_metadata(&mut self, metadata: ImageMetadata) -> Result<(), ImageError> { + metadata.validate::()?; + self.metadata = metadata; + Ok(()) + } + /// Reborrows this mutable view as read-only. #[must_use] pub fn as_view(&self) -> ImageView<'_, T, CHANNELS> { diff --git a/crates/spatialrust-platform/src/stability.rs b/crates/spatialrust-platform/src/stability.rs index 4763b08..b9387db 100644 --- a/crates/spatialrust-platform/src/stability.rs +++ b/crates/spatialrust-platform/src/stability.rs @@ -93,6 +93,10 @@ impl StabilityRegistry { "spatialrust-vision::BorderMode", "spatialrust-vision::Interpolation", "spatialrust-vision::resize", + "spatialrust-vision::resize_into", + "spatialrust-vision::normalize_into", + "spatialrust-vision::pack_chw_into", + "spatialrust-vision::rgb_to_gray_into", "spatialrust-vision::Kernel1D", "spatialrust-vision::Kernel2D", "spatialrust-vision::filter2d", @@ -172,7 +176,7 @@ mod tests { registry.lookup("spatialrust-gpu::GpuImage").unwrap().class, ApiStabilityClass::Provisional ); - assert!(registry.items().len() >= 35); + assert!(registry.items().len() >= 39); assert_eq!(registry.experimental_count(), 0); } } diff --git a/crates/spatialrust-py/spatialrust.pyi b/crates/spatialrust-py/spatialrust.pyi index 9bba7a4..381e40d 100644 --- a/crates/spatialrust-py/spatialrust.pyi +++ b/crates/spatialrust-py/spatialrust.pyi @@ -220,6 +220,7 @@ def resize_image( width: int, height: int, interpolation: str = ..., + out: Optional[_U8Array] = ..., ) -> _U8Array: ... def letterbox_image( image: _U8Array, @@ -233,8 +234,9 @@ def normalize_image_chw( scale: float = ..., mean: Optional[tuple[float, float, float]] = ..., std: Optional[tuple[float, float, float]] = ..., + out: Optional[_F32Array] = ..., ) -> _F32Array: ... -def rgb_to_gray_image(image: _U8Array) -> _U8Array: ... +def rgb_to_gray_image(image: _U8Array, out: Optional[_U8Array] = ...) -> _U8Array: ... def rgb_to_hsv_image(image: _U8Array) -> _U8Array: ... def remap_image( image: _U8Array, diff --git a/crates/spatialrust-py/src/lib.rs b/crates/spatialrust-py/src/lib.rs index ebc4e66..55fe2dd 100644 --- a/crates/spatialrust-py/src/lib.rs +++ b/crates/spatialrust-py/src/lib.rs @@ -78,22 +78,28 @@ use spatialrust::vision::{ laplacian as laplacian_op, letterbox as letterbox_op, match_descriptors as match_descriptors_op, median_blur as median_blur_op, morphology_ex as morphology_ex_op, nms as nms_op, otsu_threshold_u8 as otsu_threshold_u8_op, - pack_chw as pack_chw_op, point_map_to_point_cloud as point_map_to_cloud, - pyr_down as pyr_down_op, pyr_up as pyr_up_op, remap as remap_op, resize as resize_op, - rgb_to_gray as rgb_to_gray_op, rgb_to_hsv as rgb_to_hsv_op, scharr as scharr_op, - sobel as sobel_op, soft_nms as soft_nms_op, solve_pnp as solve_pnp_op, - stereo_block_match as stereo_block_match_op, threshold as threshold_op, - AdaptiveThresholdMethod, BinaryMask, BorderMode, BoundingBox2, CameraMatrix3, CannyOptions, - ConfidenceMap, Connectivity, CornerSelectionOptions, DescriptorBuffer, FastOptions, - HarrisOptions, Interpolation, Kernel2D, Keypoint2, MaskRle, MatchOptions, - MorphologyOperation, MorphologyShape, ObjectImageCorrespondence, OrbOptions, OrbScoreType, - PointCorrespondence2, PointMap, RobustEstimationOptions, RleOrder, ShiTomasiOptions, - SoftNmsMethod, StereoBmOptions, StructuringElement, ThresholdType, AbsolutePose, + pack_chw as pack_chw_op, pack_chw_into as pack_chw_into_op, + point_map_to_point_cloud as point_map_to_cloud, pyr_down as pyr_down_op, pyr_up as pyr_up_op, + remap as remap_op, resize as resize_op, resize_into as resize_into_op, + rgb_to_gray as rgb_to_gray_op, rgb_to_gray_into as rgb_to_gray_into_op, + rgb_to_hsv as rgb_to_hsv_op, scharr as scharr_op, sobel as sobel_op, soft_nms as soft_nms_op, + solve_pnp as solve_pnp_op, stereo_block_match as stereo_block_match_op, + threshold as threshold_op, AbsolutePose, AdaptiveThresholdMethod, BinaryMask, BorderMode, + BoundingBox2, CameraMatrix3, CannyOptions, ConfidenceMap, Connectivity, CornerSelectionOptions, + DescriptorBuffer, FastOptions, HarrisOptions, Interpolation, Kernel2D, Keypoint2, MaskRle, + MatchOptions, MorphologyOperation, MorphologyShape, ObjectImageCorrespondence, OrbOptions, + OrbScoreType, PointCorrespondence2, PointMap, RleOrder, RobustEstimationOptions, + ShiTomasiOptions, SoftNmsMethod, StereoBmOptions, StructuringElement, ThresholdType, }; use spatialrust::voxelize::{ range_image as range_image_proj, voxelize as voxelize_grid, RangeImageConfig, VoxelFill, VoxelGridConfig, }; +use spatialrust::{ + depth_to_xyz_dense as depth_to_xyz_native, depth_to_xyz_dense_into, + rgbd_to_point_cloud as rgbd_to_cloud, BrownConrady, CameraIntrinsics, DepthConversionOptions, + Image, ImageView, ImageViewMut, PinholeCamera, +}; use spatialrust::{ knn_graph as knn_graph_build, radius_graph as radius_graph_build, NeighborGraph, }; @@ -101,10 +107,6 @@ use spatialrust::{ read_point_cloud_file, write_point_cloud_file, ExecutionPolicy, HasPositions3, PointCloud, StandardSchemas, }; -use spatialrust::{ - depth_to_xyz_dense as depth_to_xyz_native, depth_to_xyz_dense_into, rgbd_to_point_cloud as rgbd_to_cloud, - BrownConrady, CameraIntrinsics, DepthConversionOptions, Image, ImageView, PinholeCamera, -}; type Vec3Tuple = (f32, f32, f32); type OrientedBoundingBoxTuple = (Vec3Tuple, Vec3Tuple, Vec); @@ -667,6 +669,24 @@ fn rgb_image_from_numpy(array: PyReadonlyArray3<'_, u8>) -> PyResult( + array: &'a PyReadonlyArray3<'py, u8>, + packed: &'a mut Vec, +) -> PyResult> { + let shape = array.shape(); + if shape.len() != 3 || shape[2] != 3 { + return Err(PyValueError::new_err("expected an (H, W, 3) uint8 RGB array")); + } + let (height, width) = (shape[0], shape[1]); + let data = if let Ok(slice) = array.as_slice() { + slice + } else { + packed.extend(array.as_array().iter().copied()); + packed.as_slice() + }; + ImageView::new(width, height, width * 3, data).map_err(to_py_err) +} + fn gray_u8_image_from_numpy(array: PyReadonlyArray2<'_, u8>) -> PyResult> { let view = array.as_array(); let shape = view.shape(); @@ -1754,11 +1774,7 @@ fn depth_to_xyz<'py>( if let Some((k1, k2, p1, p2, k3)) = distortion { camera = camera.with_distortion(BrownConrady { k1, k2, p1, p2, k3 }); } - let options = DepthConversionOptions { - depth_scale, - min_depth, - max_depth, - }; + let options = DepthConversionOptions { depth_scale, min_depth, max_depth }; if let Some(out) = out { { let mut out_rw = out.readwrite(); @@ -1850,11 +1866,7 @@ fn rgbd_to_point_cloud( if let Some((k1, k2, p1, p2, k3)) = distortion { camera = camera.with_distortion(BrownConrady { k1, k2, p1, p2, k3 }); } - let options = DepthConversionOptions { - depth_scale, - min_depth, - max_depth, - }; + let options = DepthConversionOptions { depth_scale, min_depth, max_depth }; let inner = rgbd_to_cloud(depth_image, color_image, &camera, options).map_err(to_py_err)?; Ok(PyPointCloud { inner }) } @@ -2382,12 +2394,8 @@ fn orb_features<'py>( }, ) .map_err(to_py_err)?; - let keypoints = features - .keypoints() - .iter() - .copied() - .map(|inner| PyKeypoint2 { inner }) - .collect(); + let keypoints = + features.keypoints().iter().copied().map(|inner| PyKeypoint2 { inner }).collect(); let descriptors = Array2::from_shape_vec( (features.descriptors().len(), features.descriptors().width()), features.descriptors().binary_data().expect("ORB descriptors are binary").to_vec(), @@ -2424,9 +2432,7 @@ fn mat3_to_numpy<'py>(py: Python<'py>, matrix: Mat3) -> Bound<'py, PyArray2 for row in &matrix.m { values.extend_from_slice(row); } - Array2::from_shape_vec((3, 3), values) - .expect("3x3") - .into_pyarray_bound(py) + Array2::from_shape_vec((3, 3), values).expect("3x3").into_pyarray_bound(py) } /// Estimates a homography with deterministic RANSAC. @@ -2542,21 +2548,13 @@ fn descriptor_match_tuples( ratio: Option, max_distance: Option, ) -> PyResult> { - Ok(match_descriptors_op( - &query, - &train, - MatchOptions { cross_check, ratio, max_distance }, - ) - .map_err(to_py_err)? - .into_iter() - .map(|feature_match| { - ( - feature_match.query_index(), - feature_match.train_index(), - feature_match.distance(), - ) - }) - .collect()) + Ok(match_descriptors_op(&query, &train, MatchOptions { cross_check, ratio, max_distance }) + .map_err(to_py_err)? + .into_iter() + .map(|feature_match| { + (feature_match.query_index(), feature_match.train_index(), feature_match.distance()) + }) + .collect()) } /// Brute-force Hamming matching for two `uint8[N, D]` descriptor matrices. @@ -2615,17 +2613,40 @@ fn match_float_descriptors( /// Resizes an `(H, W, 3)` uint8 RGB image. #[pyfunction] -#[pyo3(signature = (image, width, height, interpolation="bilinear"))] +#[pyo3(signature = (image, width, height, interpolation="bilinear", out=None))] fn resize_image<'py>( py: Python<'py>, image: PyReadonlyArray3<'_, u8>, width: usize, height: usize, interpolation: &str, + out: Option>>, ) -> PyResult>> { - let image = rgb_image_from_numpy(image)?; - let output = resize_op(image.view(), width, height, parse_interpolation(interpolation)?) - .map_err(to_py_err)?; + let mut packed = Vec::new(); + let image = rgb_image_view_from_numpy(&image, &mut packed)?; + let interpolation = parse_interpolation(interpolation)?; + if let Some(out) = out { + { + let mut out_rw = out.readwrite(); + let mut out_array = out_rw.as_array_mut(); + if out_array.shape() != [height, width, 3] { + return Err(PyValueError::new_err(format!( + "out shape must be ({height}, {width}, 3), found {:?}", + out_array.shape() + ))); + } + let Some(out_slice) = out_array.as_slice_mut() else { + return Err(PyValueError::new_err( + "out must be a contiguous uint8 array of shape (H, W, 3)", + )); + }; + let output = ImageViewMut::::new(width, height, width * 3, out_slice) + .map_err(to_py_err)?; + resize_into_op(image, output, interpolation).map_err(to_py_err)?; + } + return Ok(out); + } + let output = resize_op(image, width, height, interpolation).map_err(to_py_err)?; let array = Array3::from_shape_vec((height, width, 3), output.into_vec()).map_err(to_py_err)?; Ok(array.into_pyarray_bound(py)) } @@ -2666,18 +2687,41 @@ fn letterbox_image<'py>( /// Normalizes RGB and packs it into a float32 `(3, H, W)` CHW tensor. #[pyfunction] -#[pyo3(signature = (image, scale=1.0/255.0, mean=None, std=None))] +#[pyo3(signature = (image, scale=1.0/255.0, mean=None, std=None, out=None))] fn normalize_image_chw<'py>( py: Python<'py>, image: PyReadonlyArray3<'_, u8>, scale: f32, mean: Option<(f32, f32, f32)>, std: Option<(f32, f32, f32)>, + out: Option>>, ) -> PyResult>> { - let image = rgb_image_from_numpy(image)?; + let mut packed = Vec::new(); + let image = rgb_image_view_from_numpy(&image, &mut packed)?; let mean = mean.map_or([0.0; 3], |(r, g, b)| [r, g, b]); let std = std.map_or([1.0; 3], |(r, g, b)| [r, g, b]); - let output = pack_chw_op(image.view(), scale, mean, std).map_err(to_py_err)?; + if let Some(out) = out { + { + let mut out_rw = out.readwrite(); + let mut out_array = out_rw.as_array_mut(); + if out_array.shape() != [3, image.height(), image.width()] { + return Err(PyValueError::new_err(format!( + "out shape must be (3, {}, {}), found {:?}", + image.height(), + image.width(), + out_array.shape() + ))); + } + let Some(out_slice) = out_array.as_slice_mut() else { + return Err(PyValueError::new_err( + "out must be a contiguous float32 array of shape (3, H, W)", + )); + }; + pack_chw_into_op(image, scale, mean, std, out_slice).map_err(to_py_err)?; + } + return Ok(out); + } + let output = pack_chw_op(image, scale, mean, std).map_err(to_py_err)?; let array = Array3::from_shape_vec((3, image.height(), image.width()), output.into_vec()) .map_err(to_py_err)?; Ok(array.into_pyarray_bound(py)) @@ -2685,12 +2729,39 @@ fn normalize_image_chw<'py>( /// Converts an RGB image to an `(H, W)` grayscale image. #[pyfunction] +#[pyo3(signature = (image, out=None))] fn rgb_to_gray_image<'py>( py: Python<'py>, image: PyReadonlyArray3<'_, u8>, + out: Option>>, ) -> PyResult>> { - let image = rgb_image_from_numpy(image)?; - let output = rgb_to_gray_op(image.view()).map_err(to_py_err)?; + let mut packed = Vec::new(); + let image = rgb_image_view_from_numpy(&image, &mut packed)?; + if let Some(out) = out { + { + let mut out_rw = out.readwrite(); + let mut out_array = out_rw.as_array_mut(); + if out_array.shape() != [image.height(), image.width()] { + return Err(PyValueError::new_err(format!( + "out shape must be ({}, {}), found {:?}", + image.height(), + image.width(), + out_array.shape() + ))); + } + let Some(out_slice) = out_array.as_slice_mut() else { + return Err(PyValueError::new_err( + "out must be a contiguous uint8 array of shape (H, W)", + )); + }; + let output = + ImageViewMut::::new(image.width(), image.height(), image.width(), out_slice) + .map_err(to_py_err)?; + rgb_to_gray_into_op(image, output).map_err(to_py_err)?; + } + return Ok(out); + } + let output = rgb_to_gray_op(image).map_err(to_py_err)?; let array = Array2::from_shape_vec((image.height(), image.width()), output.into_vec()) .map_err(to_py_err)?; Ok(array.into_pyarray_bound(py)) diff --git a/crates/spatialrust-py/tests/test_bindings.py b/crates/spatialrust-py/tests/test_bindings.py index 2bd925d..522cadc 100644 --- a/crates/spatialrust-py/tests/test_bindings.py +++ b/crates/spatialrust-py/tests/test_bindings.py @@ -134,6 +134,9 @@ def test_image_resize_letterbox_and_normalize(): resized = sr.resize_image(image, 4, 4, interpolation="nearest") assert resized.shape == (4, 4, 3) np.testing.assert_array_equal(resized[0, 0], image[0, 0]) + resized_out = np.empty((4, 4, 3), dtype=np.uint8) + assert sr.resize_image(image, 4, 4, interpolation="nearest", out=resized_out) is resized_out + np.testing.assert_array_equal(resized_out, resized) letterboxed, transform = sr.letterbox_image(image, 4, 6, fill=(7, 8, 9)) assert letterboxed.shape == (6, 4, 3) @@ -144,6 +147,9 @@ def test_image_resize_letterbox_and_normalize(): assert chw.shape == (3, 2, 2) assert chw.dtype == np.float32 np.testing.assert_allclose(chw[:, 0, 0], [1.0, 0.0, 0.0], atol=1e-6) + chw_out = np.empty((3, 2, 2), dtype=np.float32) + assert sr.normalize_image_chw(image, out=chw_out) is chw_out + np.testing.assert_allclose(chw_out, chw, atol=1e-6) def test_image_color_and_remap(): @@ -151,6 +157,9 @@ def test_image_color_and_remap(): gray = sr.rgb_to_gray_image(image) assert gray.shape == (1, 2) np.testing.assert_allclose(gray, [[76, 150]], atol=1) + gray_out = np.empty((1, 2), dtype=np.uint8) + assert sr.rgb_to_gray_image(image, out=gray_out) is gray_out + np.testing.assert_array_equal(gray_out, gray) hsv = sr.rgb_to_hsv_image(image) np.testing.assert_array_equal(hsv[0, 0], [0, 255, 255]) np.testing.assert_array_equal(hsv[0, 1], [60, 255, 255]) diff --git a/crates/spatialrust-vision/Cargo.toml b/crates/spatialrust-vision/Cargo.toml index c8e65c8..789d041 100644 --- a/crates/spatialrust-vision/Cargo.toml +++ b/crates/spatialrust-vision/Cargo.toml @@ -43,6 +43,11 @@ name = "preprocess" harness = false required-features = ["preprocess"] +[[bench]] +name = "resize" +harness = false +required-features = ["resize"] + [[bench]] name = "filter" harness = false diff --git a/crates/spatialrust-vision/benches/preprocess.rs b/crates/spatialrust-vision/benches/preprocess.rs index 0a0258d..f4d8386 100644 --- a/crates/spatialrust-vision/benches/preprocess.rs +++ b/crates/spatialrust-vision/benches/preprocess.rs @@ -1,6 +1,8 @@ -use criterion::{black_box, criterion_group, criterion_main, Criterion}; +use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; use spatialrust_image::Image; -use spatialrust_vision::{letterbox, pack_chw, Interpolation}; +use spatialrust_vision::{ + letterbox, pack_chw, pack_chw_into, rgb_to_gray, rgb_to_gray_into, Interpolation, +}; fn benchmark_preprocess(c: &mut Criterion) { let input = Image::::try_new(1280, 720, vec![127; 1280 * 720 * 3]).unwrap(); @@ -14,5 +16,39 @@ fn benchmark_preprocess(c: &mut Criterion) { }); } -criterion_group!(benches, benchmark_preprocess); +fn benchmark_reusable_preprocess(c: &mut Criterion) { + for &(name, width, height) in &[("640p", 640, 480), ("1080p", 1920, 1080), ("4k", 3840, 2160)] { + let input = Image::::try_new(width, height, vec![127; width * height * 3]).unwrap(); + let mut gray = Image::::try_new(width, height, vec![0; width * height]).unwrap(); + let mut chw = vec![0.0_f32; width * height * 3]; + let throughput = Throughput::Elements((width * height) as u64); + + let mut gray_group = c.benchmark_group("rgb_to_gray_rgb8"); + gray_group.sample_size(10); + gray_group.throughput(throughput.clone()); + gray_group.bench_function(BenchmarkId::new("allocate", name), |b| { + b.iter(|| rgb_to_gray(black_box(input.view())).unwrap()); + }); + gray_group.bench_function(BenchmarkId::new("reuse", name), |b| { + b.iter(|| rgb_to_gray_into(black_box(input.view()), gray.view_mut()).unwrap()); + }); + gray_group.finish(); + + let mut chw_group = c.benchmark_group("pack_chw_rgb8"); + chw_group.sample_size(10); + chw_group.throughput(throughput); + chw_group.bench_function(BenchmarkId::new("allocate", name), |b| { + b.iter(|| pack_chw(black_box(input.view()), 1.0 / 255.0, [0.0; 3], [1.0; 3]).unwrap()); + }); + chw_group.bench_function(BenchmarkId::new("reuse", name), |b| { + b.iter(|| { + pack_chw_into(black_box(input.view()), 1.0 / 255.0, [0.0; 3], [1.0; 3], &mut chw) + .unwrap() + }); + }); + chw_group.finish(); + } +} + +criterion_group!(benches, benchmark_preprocess, benchmark_reusable_preprocess); criterion_main!(benches); diff --git a/crates/spatialrust-vision/benches/resize.rs b/crates/spatialrust-vision/benches/resize.rs new file mode 100644 index 0000000..8b7f42d --- /dev/null +++ b/crates/spatialrust-vision/benches/resize.rs @@ -0,0 +1,41 @@ +use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; +use spatialrust_image::Image; +use spatialrust_vision::{resize, resize_into, Interpolation}; + +fn benchmark_resize(c: &mut Criterion) { + let mut group = c.benchmark_group("resize_bilinear_rgb8"); + group.sample_size(10); + for &(name, width, height) in &[("640p", 640, 480), ("1080p", 1920, 1080), ("4k", 3840, 2160)] { + let input = Image::::try_new(width, height, vec![127; width * height * 3]).unwrap(); + let output_width = width / 2; + let output_height = height / 2; + let mut output = Image::::try_new( + output_width, + output_height, + vec![0; output_width * output_height * 3], + ) + .unwrap(); + group.throughput(Throughput::Elements((output_width * output_height) as u64)); + group.bench_function(BenchmarkId::new("allocate", name), |b| { + b.iter(|| { + resize( + black_box(input.view()), + output_width, + output_height, + Interpolation::Bilinear, + ) + .unwrap() + }); + }); + group.bench_function(BenchmarkId::new("reuse", name), |b| { + b.iter(|| { + resize_into(black_box(input.view()), output.view_mut(), Interpolation::Bilinear) + .unwrap() + }); + }); + } + group.finish(); +} + +criterion_group!(benches, benchmark_resize); +criterion_main!(benches); diff --git a/crates/spatialrust-vision/src/preprocess.rs b/crates/spatialrust-vision/src/preprocess.rs index 0c8530a..1b1a9d2 100644 --- a/crates/spatialrust-vision/src/preprocess.rs +++ b/crates/spatialrust-vision/src/preprocess.rs @@ -1,4 +1,6 @@ -use spatialrust_image::{ColorSpace, Image, ImageMetadata, ImageRegion, ImageView, PlanarImage}; +use spatialrust_image::{ + ColorSpace, Image, ImageMetadata, ImageRegion, ImageView, ImageViewMut, PlanarImage, +}; use crate::{resize, Interpolation, PixelComponent, VisionError, VisionResult}; @@ -142,15 +144,10 @@ pub fn normalize( mean: [f32; CHANNELS], std: [f32; CHANNELS], ) -> VisionResult> { - if !scale.is_finite() || std.iter().any(|value| !value.is_finite() || *value == 0.0) { - return Err(VisionError::InvalidParameter( - "normalization scale/std must be finite and std non-zero".to_owned(), - )); - } + validate_normalization(scale, std)?; let mut output = Vec::with_capacity(input.width() * input.height() * CHANNELS); for y in 0..input.height() { - for x in 0..input.width() { - let pixel = input.get(x, y).expect("image coordinate in bounds"); + for pixel in input.row(y).expect("input row in bounds").chunks_exact(CHANNELS) { for channel in 0..CHANNELS { output .push((pixel[channel].to_f64() as f32 * scale - mean[channel]) / std[channel]); @@ -160,6 +157,32 @@ pub fn normalize( Ok(Image::try_new_with_metadata(input.width(), input.height(), output, input.metadata())?) } +/// Normalizes interleaved pixels into caller-owned `f32` storage. +pub fn normalize_into( + input: ImageView<'_, T, CHANNELS>, + mut output: ImageViewMut<'_, f32, CHANNELS>, + scale: f32, + mean: [f32; CHANNELS], + std: [f32; CHANNELS], +) -> VisionResult<()> { + validate_normalization(scale, std)?; + validate_output_dimensions(input, output.width(), output.height())?; + output.set_metadata(input.metadata())?; + for y in 0..input.height() { + let source = input.row(y).expect("input row in bounds"); + let target = output.row_mut(y).expect("output row in bounds"); + for (source_pixel, target_pixel) in + source.chunks_exact(CHANNELS).zip(target.chunks_exact_mut(CHANNELS)) + { + for channel in 0..CHANNELS { + target_pixel[channel] = + (source_pixel[channel].to_f64() as f32 * scale - mean[channel]) / std[channel]; + } + } + } + Ok(()) +} + /// Packs and normalizes an interleaved image into planar CHW storage. pub fn pack_chw( input: ImageView<'_, T, CHANNELS>, @@ -167,24 +190,97 @@ pub fn pack_chw( mean: [f32; CHANNELS], std: [f32; CHANNELS], ) -> VisionResult> { + let plane_len = input.width() * input.height(); + let mut output = vec![0.0_f32; plane_len * CHANNELS]; + pack_chw_into(input, scale, mean, std, &mut output)?; + Ok(PlanarImage::try_new_with_metadata(input.width(), input.height(), output, input.metadata())?) +} + +/// Packs and normalizes interleaved pixels into a reusable planar CHW slice. +/// +/// Large multi-channel inputs dispatch across channels with safe scoped +/// threads. Smaller inputs remain scalar to avoid scheduling overhead. +pub fn pack_chw_into( + input: ImageView<'_, T, CHANNELS>, + scale: f32, + mean: [f32; CHANNELS], + std: [f32; CHANNELS], + output: &mut [f32], +) -> VisionResult<()> { + validate_normalization(scale, std)?; + let plane_len = input.width().checked_mul(input.height()).ok_or_else(|| { + VisionError::InvalidDimensions("CHW plane dimensions overflow".to_owned()) + })?; + let required = plane_len.checked_mul(CHANNELS).ok_or_else(|| { + VisionError::InvalidDimensions("CHW output dimensions overflow".to_owned()) + })?; + if output.len() != required { + return Err(VisionError::ShapeMismatch(format!( + "CHW output needs {required} elements, found {}", + output.len() + ))); + } + if plane_len == 0 { + return Ok(()); + } + const PARALLEL_PLANE_THRESHOLD: usize = 256 * 1024; + if CHANNELS > 1 && plane_len >= PARALLEL_PLANE_THRESHOLD { + std::thread::scope(|scope| { + for (channel, plane) in output.chunks_exact_mut(plane_len).enumerate() { + scope.spawn(move || { + fill_chw_plane(input, channel, scale, mean[channel], std[channel], plane) + }); + } + }); + } else { + for (channel, plane) in output.chunks_exact_mut(plane_len).enumerate() { + fill_chw_plane(input, channel, scale, mean[channel], std[channel], plane); + } + } + Ok(()) +} + +fn fill_chw_plane( + input: ImageView<'_, T, CHANNELS>, + channel: usize, + scale: f32, + mean: f32, + std: f32, + output: &mut [f32], +) { + for y in 0..input.height() { + let row = input.row(y).expect("input row in bounds"); + for (x, pixel) in row.chunks_exact(CHANNELS).enumerate() { + output[y * input.width() + x] = (pixel[channel].to_f64() as f32 * scale - mean) / std; + } + } +} + +fn validate_normalization( + scale: f32, + std: [f32; CHANNELS], +) -> VisionResult<()> { if !scale.is_finite() || std.iter().any(|value| !value.is_finite() || *value == 0.0) { return Err(VisionError::InvalidParameter( "normalization scale/std must be finite and std non-zero".to_owned(), )); } - let plane_len = input.width() * input.height(); - let mut output = vec![0.0_f32; plane_len * CHANNELS]; - for y in 0..input.height() { - for x in 0..input.width() { - let pixel = input.get(x, y).expect("image coordinate in bounds"); - let index = y * input.width() + x; - for channel in 0..CHANNELS { - output[channel * plane_len + index] = - (pixel[channel].to_f64() as f32 * scale - mean[channel]) / std[channel]; - } - } + Ok(()) +} + +fn validate_output_dimensions( + input: ImageView<'_, T, CHANNELS>, + output_width: usize, + output_height: usize, +) -> VisionResult<()> { + if (input.width(), input.height()) != (output_width, output_height) { + return Err(VisionError::ShapeMismatch(format!( + "output dimensions {output_width}x{output_height} do not match input {}x{}", + input.width(), + input.height() + ))); } - Ok(PlanarImage::try_new_with_metadata(input.width(), input.height(), output, input.metadata())?) + Ok(()) } /// Swaps the red and blue channels of a three-channel image. @@ -209,20 +305,39 @@ pub fn swap_red_blue(input: ImageView<'_, T, 3>) -> VisionRes pub fn rgb_to_gray(input: ImageView<'_, u8, 3>) -> VisionResult> { let mut data = Vec::with_capacity(input.width() * input.height()); for y in 0..input.height() { - for x in 0..input.width() { - let pixel = input.get(x, y).expect("image coordinate in bounds"); - let value = (77_u32 * u32::from(pixel[0]) - + 150_u32 * u32::from(pixel[1]) - + 29_u32 * u32::from(pixel[2]) - + 128) - >> 8; - data.push(value as u8); + for pixel in input.row(y).expect("input row in bounds").chunks_exact(3) { + data.push(rgb_luma(pixel)); } } let metadata = ImageMetadata { color_space: ColorSpace::Gray, ..input.metadata() }; Ok(Image::try_new_with_metadata(input.width(), input.height(), data, metadata)?) } +/// Converts RGB `u8` pixels into caller-owned gray storage without allocating. +pub fn rgb_to_gray_into( + input: ImageView<'_, u8, 3>, + mut output: ImageViewMut<'_, u8, 1>, +) -> VisionResult<()> { + validate_output_dimensions(input, output.width(), output.height())?; + output.set_metadata(ImageMetadata { color_space: ColorSpace::Gray, ..input.metadata() })?; + for y in 0..input.height() { + let source = input.row(y).expect("input row in bounds"); + let target = output.row_mut(y).expect("output row in bounds"); + for (pixel, target_value) in source.chunks_exact(3).zip(target.iter_mut()) { + *target_value = rgb_luma(pixel); + } + } + Ok(()) +} + +fn rgb_luma(pixel: &[u8]) -> u8 { + ((77_u32 * u32::from(pixel[0]) + + 150_u32 * u32::from(pixel[1]) + + 29_u32 * u32::from(pixel[2]) + + 128) + >> 8) as u8 +} + /// Replicates a gray channel into RGB. pub fn gray_to_rgb(input: ImageView<'_, T, 1>) -> VisionResult> { let mut data = Vec::with_capacity(input.width() * input.height() * 3); @@ -275,10 +390,11 @@ pub fn rgb_to_hsv(input: ImageView<'_, u8, 3>) -> VisionResult> { #[cfg(test)] mod tests { use super::{ - crop, gray_to_rgb, letterbox, normalize, pack_chw, pad, rgb_to_gray, rgb_to_hsv, Padding, + crop, gray_to_rgb, letterbox, normalize, normalize_into, pack_chw, pack_chw_into, pad, + rgb_to_gray, rgb_to_gray_into, rgb_to_hsv, Padding, }; use crate::Interpolation; - use spatialrust_image::{ColorSpace, Image, ImageMetadata, ImageRegion}; + use spatialrust_image::{ColorSpace, Image, ImageMetadata, ImageRegion, ImageViewMut}; #[test] fn crop_and_pad_roundtrip_center() { @@ -311,6 +427,48 @@ mod tests { assert_eq!(chw.as_slice(), &[0.0, 3.0, 1.0, 4.0, 2.0, 5.0]); } + #[test] + fn reusable_normalize_and_chw_match_owned_outputs() { + let image = Image::::try_new(2, 1, vec![0, 10, 20, 30, 40, 50]).unwrap(); + let mut interleaved = vec![-1.0_f32; 9]; + let output = ImageViewMut::::new(2, 1, 9, &mut interleaved).unwrap(); + normalize_into(image.view(), output, 0.1, [0.0; 3], [1.0; 3]).unwrap(); + assert_eq!( + &interleaved[..6], + normalize(image.view(), 0.1, [0.0; 3], [1.0; 3]).unwrap().as_slice() + ); + assert_eq!(&interleaved[6..], &[-1.0; 3]); + + let mut chw = vec![0.0; 6]; + pack_chw_into(image.view(), 0.1, [0.0; 3], [1.0; 3], &mut chw).unwrap(); + assert_eq!(chw, pack_chw(image.view(), 0.1, [0.0; 3], [1.0; 3]).unwrap().as_slice()); + assert!(pack_chw_into(image.view(), 1.0, [0.0; 3], [1.0; 3], &mut chw[..5]).is_err()); + + let empty = Image::::try_new(0, 0, Vec::new()).unwrap(); + pack_chw_into(empty.view(), 1.0, [0.0; 3], [1.0; 3], &mut []).unwrap(); + } + + #[test] + fn large_chw_parallel_dispatch_matches_channel_layout() { + let image = Image::::try_new(512, 512, vec![10; 512 * 512 * 3]).unwrap(); + let mut chw = vec![0.0; 512 * 512 * 3]; + pack_chw_into(image.view(), 0.1, [0.0, 0.5, 1.0], [1.0; 3], &mut chw).unwrap(); + let plane = 512 * 512; + assert!(chw[..plane].iter().all(|&value| value == 1.0)); + assert!(chw[plane..2 * plane].iter().all(|&value| value == 0.5)); + assert!(chw[2 * plane..].iter().all(|&value| value == 0.0)); + } + + #[test] + fn rgb_to_gray_into_accepts_strided_output() { + let image = Image::::try_new(2, 1, vec![255, 0, 0, 0, 255, 0]).unwrap(); + let mut storage = vec![200_u8; 4]; + let output = ImageViewMut::::new(2, 1, 4, &mut storage).unwrap(); + rgb_to_gray_into(image.view(), output).unwrap(); + assert_eq!(&storage[..2], rgb_to_gray(image.view()).unwrap().as_slice()); + assert_eq!(&storage[2..], &[200, 200]); + } + #[test] fn color_conversions_match_known_primaries() { let metadata = ImageMetadata { color_space: ColorSpace::Rgb, ..Default::default() }; diff --git a/crates/spatialrust-vision/src/resize.rs b/crates/spatialrust-vision/src/resize.rs index 0a5fa27..db8146e 100644 --- a/crates/spatialrust-vision/src/resize.rs +++ b/crates/spatialrust-vision/src/resize.rs @@ -1,4 +1,4 @@ -use spatialrust_image::{Image, ImageView}; +use spatialrust_image::{Image, ImageView, ImageViewMut}; use crate::{PixelComponent, VisionError, VisionResult}; @@ -43,23 +43,84 @@ pub fn resize( && (output_width < input.width() || output_height < input.height()); for y in 0..output_height { for x in 0..output_width { - let pixel = if area_downsample { - sample_area(input, x, y, output_width, output_height) - } else { - let sx = half_pixel_coordinate(x, input.width(), output_width); - let sy = half_pixel_coordinate(y, input.height(), output_height); - match interpolation { - Interpolation::Nearest => sample_nearest(input, sx, sy), - Interpolation::Bilinear | Interpolation::Area => sample_bilinear(input, sx, sy), - Interpolation::Bicubic => sample_bicubic(input, sx, sy), - } - }; + let pixel = resized_pixel( + input, + x, + y, + output_width, + output_height, + interpolation, + area_downsample, + ); output.extend_from_slice(&pixel); } } Ok(Image::try_new_with_metadata(output_width, output_height, output, input.metadata())?) } +/// Resizes into caller-owned storage without allocating. +/// +/// The output dimensions select the requested size. Packed and strided output +/// views are accepted, and semantic metadata is replaced with the input +/// metadata after channel validation. +pub fn resize_into( + input: ImageView<'_, T, CHANNELS>, + mut output: ImageViewMut<'_, T, CHANNELS>, + interpolation: Interpolation, +) -> VisionResult<()> { + let output_width = output.width(); + let output_height = output.height(); + output.set_metadata(input.metadata())?; + if output_width == 0 || output_height == 0 { + return Ok(()); + } + if input.width() == 0 || input.height() == 0 { + return Err(VisionError::InvalidDimensions( + "cannot resize an empty input to a non-empty output".to_owned(), + )); + } + let area_downsample = interpolation == Interpolation::Area + && (output_width < input.width() || output_height < input.height()); + for y in 0..output_height { + let row = output.row_mut(y).expect("output row in bounds"); + for x in 0..output_width { + let pixel = resized_pixel( + input, + x, + y, + output_width, + output_height, + interpolation, + area_downsample, + ); + row[x * CHANNELS..(x + 1) * CHANNELS].copy_from_slice(&pixel); + } + } + Ok(()) +} + +fn resized_pixel( + input: ImageView<'_, T, CHANNELS>, + x: usize, + y: usize, + output_width: usize, + output_height: usize, + interpolation: Interpolation, + area_downsample: bool, +) -> [T; CHANNELS] { + if area_downsample { + sample_area(input, x, y, output_width, output_height) + } else { + let sx = half_pixel_coordinate(x, input.width(), output_width); + let sy = half_pixel_coordinate(y, input.height(), output_height); + match interpolation { + Interpolation::Nearest => sample_nearest(input, sx, sy), + Interpolation::Bilinear | Interpolation::Area => sample_bilinear(input, sx, sy), + Interpolation::Bicubic => sample_bicubic(input, sx, sy), + } + } +} + fn half_pixel_coordinate(output: usize, input_len: usize, output_len: usize) -> f64 { (output as f64 + 0.5) * input_len as f64 / output_len as f64 - 0.5 } @@ -173,8 +234,8 @@ fn sample_area( #[cfg(test)] mod tests { - use super::{resize, Interpolation}; - use spatialrust_image::Image; + use super::{resize, resize_into, Interpolation}; + use spatialrust_image::{Image, ImageViewMut}; #[test] fn nearest_repeats_pixels() { @@ -209,4 +270,19 @@ mod tests { assert_eq!(resize(input.view(), 2, 2, filter).unwrap(), input); } } + + #[test] + fn resize_into_reuses_strided_output_and_preserves_padding() { + let input = Image::::try_new(2, 2, vec![0, 10, 20, 30]).unwrap(); + let mut storage = vec![99_u8; 15]; + let output = ImageViewMut::::new(3, 3, 6, &mut storage).unwrap(); + resize_into(input.view(), output, Interpolation::Bilinear).unwrap(); + let expected = resize(input.view(), 3, 3, Interpolation::Bilinear).unwrap(); + for y in 0..3 { + assert_eq!(&storage[y * 6..y * 6 + 3], expected.view().row(y).unwrap()); + if y < 2 { + assert_eq!(&storage[y * 6 + 3..(y + 1) * 6], &[99; 3]); + } + } + } } diff --git a/crates/spatialrust/tests/vision_api_v1.rs b/crates/spatialrust/tests/vision_api_v1.rs index 7957aa4..517deae 100644 --- a/crates/spatialrust/tests/vision_api_v1.rs +++ b/crates/spatialrust/tests/vision_api_v1.rs @@ -1,10 +1,11 @@ //! Compile-and-behavior contract for the stable SpatialRust Vision 1.x entry surface. use spatialrust::camera::{CameraIntrinsics, PinholeCamera}; -use spatialrust::image::{ColorSpace, Image, ImageMetadata, ImageRegion}; +use spatialrust::image::{ColorSpace, Image, ImageMetadata, ImageRegion, ImageViewMut}; use spatialrust::platform::{ApiStabilityClass, StabilityRegistry}; use spatialrust::vision::{ - filter2d, nms, resize, BorderMode, BoundingBox2, Interpolation, Kernel2D, + filter2d, nms, normalize_into, pack_chw_into, resize, resize_into, rgb_to_gray_into, + BorderMode, BoundingBox2, Interpolation, Kernel2D, }; #[test] @@ -18,6 +19,21 @@ fn stable_image_camera_and_vision_entry_points_compose() { assert_eq!((filtered.width(), filtered.height()), (4, 4)); assert_eq!(filtered.metadata(), metadata); + let rgb = Image::::try_new(2, 1, vec![255, 0, 0, 0, 255, 0]).unwrap(); + let mut resized_storage = vec![0_u8; 4 * 2 * 3]; + let resized_output = ImageViewMut::new(4, 2, 4 * 3, &mut resized_storage).unwrap(); + resize_into(rgb.view(), resized_output, Interpolation::Nearest).unwrap(); + let mut gray_storage = vec![0_u8; 2]; + let gray_output = ImageViewMut::new(2, 1, 2, &mut gray_storage).unwrap(); + rgb_to_gray_into(rgb.view(), gray_output).unwrap(); + let mut normalized_storage = vec![0.0_f32; 6]; + let normalized_output = ImageViewMut::new(2, 1, 6, &mut normalized_storage).unwrap(); + normalize_into(rgb.view(), normalized_output, 1.0 / 255.0, [0.0; 3], [1.0; 3]).unwrap(); + let mut chw = vec![0.0_f32; 6]; + pack_chw_into(rgb.view(), 1.0 / 255.0, [0.0; 3], [1.0; 3], &mut chw).unwrap(); + assert_eq!(gray_storage, [77, 149]); + assert_eq!(chw, [1.0, 0.0, 0.0, 1.0, 0.0, 0.0]); + let intrinsics = CameraIntrinsics::try_new(100.0, 100.0, 1.5, 1.0, 4, 3).unwrap(); let camera = PinholeCamera::new(intrinsics); let pixel = spatialrust::Vec2 { x: 1.5, y: 1.0 }; @@ -35,4 +51,8 @@ fn stable_image_camera_and_vision_entry_points_compose() { registry.lookup("spatialrust-vision::resize").unwrap().class, ApiStabilityClass::Stable ); + assert_eq!( + registry.lookup("spatialrust-vision::pack_chw_into").unwrap().class, + ApiStabilityClass::Stable + ); } diff --git a/docs/API_STABILITY.md b/docs/API_STABILITY.md index 378e5c0..368cfcd 100644 --- a/docs/API_STABILITY.md +++ b/docs/API_STABILITY.md @@ -105,7 +105,7 @@ spatialrust- / feature- | `spatialrust-distribute` | Provisional | Partition graphs, topo schedules, backpressure queues, named measurable transfers | | `spatialrust-platform` | Provisional | Stability registry, conformance summaries, security checklist, LTS policy, performance budgets, release gate | | `spatialrust-camera` | Stable foundation | Pinhole/Brown–Conrady and named RGB-D conversion entry points; future calibration solvers are additive and provisional | -| `spatialrust-vision` | Stable foundation | Errors, borders, resize/filter entry points, detection/dense and Feature2D data contracts are stable; geometry/stereo/flow/AI adapters remain provisional | +| `spatialrust-vision` | Stable foundation | Errors, borders, resize/filter entry points, reusable resize/gray/normalize/CHW outputs, detection/dense and Feature2D data contracts are stable; geometry/stereo/flow/AI adapters remain provisional | | `spatialrust-search` | Stable with features | KD-tree behind `search-kdtree`; **chunked query traits** and **`search-parallel`** provisional | | `spatialrust-filtering` | Provisional | GPU thresholds may move | | `spatialrust-features` | Provisional | Normal GPU path still tuning | diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 722446a..95369b6 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -378,7 +378,7 @@ uses the standard completion gates above and lands as one reviewable PR. | --- | --- | --- | --- | | 101 | Complete | 83–90 | Reproducible OpenCV correctness/performance contract, workload manifest, environment receipts, and aggregate runner | | 102 | Complete | 101 | Stabilize the image/camera/vision 1.0 contract and cross-platform conformance | -| 103 | Planned | 101–102 | SIMD/parallel CPU kernel dispatch, reusable outputs, and measured allocation control | +| 103 | Complete | 101–102 | SIMD/parallel CPU kernel dispatch, reusable outputs, and measured allocation control | | 104 | Planned | 89, 101–103 | Texture-backed GPU Image v2 and device-resident resize/filter/edge/morphology chains | | 105 | Planned | 88, 101–102 | Mono/stereo/fisheye/hand-eye calibration and bundle-adjustment contracts | | 106 | Planned | 92, 101–105 | Dense flow, tracking, background modeling, and feature-gated video stream adapters | @@ -422,3 +422,19 @@ and `GpuImage` remain explicitly provisional. Stable entries may gain faster internal dispatch without changing ownership, error, stride, or transfer semantics. Completion requires the dedicated three-OS CI matrix and the full `spatialrust-vision/full` property suite to pass. + +### Epic 103 delivery slices + +| Slice | Status | Scope | Evidence | +| --- | --- | --- | --- | +| 103A | Complete | Allocation audit and caller-owned image/planar outputs | `resize_into`, `rgb_to_gray_into`, `normalize_into`, `pack_chw_into` | +| 103B | Complete | Safe size-aware parallel dispatch for planar AI packing | scoped channel workers with scalar small-image fallback | +| 103C | Complete | Rust and NumPy reusable-output contracts | strided Rust tests and Python `out=` identity tests | +| 103D | Complete | VGA/1080p/4K correctness and allocation/reuse measurements against OpenCV | `opencv-vision-performance` report | + +Epic 103 does not claim blanket CPU kernel superiority. The comparison receipt +records OpenCV's SIMD advantage for resize and RGB-to-gray, while SpatialRust's +typed RGB-to-CHW path is faster on every canonical profile. On the reference +Windows host, reusable SpatialRust CHW measured 8.54x, 13.07x, and 16.11x faster +than allocating `cv2.dnn.blobFromImage` at VGA, 1080p, and 4K. Public CPU APIs +accept explicit caller storage and never perform an implicit device transfer. diff --git a/notes/2026-07-15_epic103_cpu_vision.md b/notes/2026-07-15_epic103_cpu_vision.md new file mode 100644 index 0000000..d9e2772 --- /dev/null +++ b/notes/2026-07-15_epic103_cpu_vision.md @@ -0,0 +1,20 @@ +# Epic 103: reusable CPU vision paths + +Epic 103 adds explicit caller-owned output paths for resize, RGB-to-gray, +interleaved normalization, and planar CHW packing. Packed and strided image +views remain safe, metadata is propagated deliberately, and no CPU API performs +an implicit device transfer. + +Large CHW buffers dispatch one safe scoped worker per channel; smaller inputs +stay scalar to avoid thread overhead. Python exposes the same ownership choice +through optional NumPy `out=` arrays and borrows contiguous RGB input without a +packing copy. + +The canonical OpenCV performance receipt was recorded on Windows 11, CPython +3.12.10, OpenCV 4.10.0, and a 12-logical-CPU Intel host. Correctness stayed +within one `u8` value for resize/gray and `5.97e-8` for normalized CHW. OpenCV +remained substantially faster for bilinear resize and RGB-to-gray. SpatialRust +was 3.97x/5.76x/7.62x faster for allocating RGB-to-CHW and +8.54x/13.07x/16.11x faster when reusing output at VGA/1080p/4K respectively. +Raw samples, medians, p95, min/max, and environment details are emitted by +`bench/opencv_vision_comparison/performance.py` rather than checked into source.