From 71c469dc79bd70c05515433f13e028ccd9017634 Mon Sep 17 00:00:00 2001 From: rsasaki0109 Date: Wed, 15 Jul 2026 17:33:44 +0900 Subject: [PATCH] Add Epic 108 computational photography --- CHANGELOG.md | 5 + bench/opencv_comparison/manifest.json | 2 + bench/opencv_comparison/run.py | 1 + bench/opencv_photography_comparison/README.md | 4 + bench/opencv_photography_comparison/run.py | 61 ++++ crates/spatialrust-platform/src/stability.rs | 1 + crates/spatialrust-py/spatialrust.pyi | 10 +- crates/spatialrust-py/src/lib.rs | 67 +++- crates/spatialrust-vision/Cargo.toml | 8 +- .../spatialrust-vision/benches/photography.rs | 21 ++ crates/spatialrust-vision/src/lib.rs | 4 + crates/spatialrust-vision/src/photography.rs | 303 ++++++++++++++++++ crates/spatialrust/Cargo.toml | 2 + docs/API_STABILITY.md | 2 +- docs/ARCHITECTURE.md | 3 + docs/ROADMAP.md | 17 +- notes/2026-07-15_epic108_photography.md | 11 + 17 files changed, 511 insertions(+), 11 deletions(-) create mode 100644 bench/opencv_photography_comparison/README.md create mode 100644 bench/opencv_photography_comparison/run.py create mode 100644 crates/spatialrust-vision/benches/photography.rs create mode 100644 crates/spatialrust-vision/src/photography.rs create mode 100644 notes/2026-07-15_epic108_photography.md diff --git a/CHANGELOG.md b/CHANGELOG.md index eaf8512..a78c2b5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,6 +21,11 @@ removed no sooner than the next major (see `docs/API_STABILITY.md`). ### Added +- **Computational photography and panorama (Epic 108)**: gray-world white + balance, aligned well-exposedness fusion, bounded homography panorama canvas, + feather blending, Python entry points, and OpenCV warp parity receipts behind + an additive vision feature. + - **Robust visual/RGB-D odometry integration (Epic 107)**: spatially balanced keypoint selection, forward/backward LK validation, scale-explicit monocular motion, metric depth-backed PnP, mapping `DeltaMotion` bridges, Python RGB-D diff --git a/bench/opencv_comparison/manifest.json b/bench/opencv_comparison/manifest.json index 74c90b8..bb63a1b 100644 --- a/bench/opencv_comparison/manifest.json +++ b/bench/opencv_comparison/manifest.json @@ -30,6 +30,8 @@ { "id": "multi_object_tracking", "domain": "video", "modes": ["streaming"] }, { "id": "visual_odometry", "domain": "localization", "modes": ["correctness"] }, { "id": "rgbd_odometry", "domain": "localization", "modes": ["correctness"] }, + { "id": "exposure_fusion", "domain": "photography", "modes": ["allocate"] }, + { "id": "panorama_pair", "domain": "photography", "modes": ["allocate"] }, { "id": "depth_to_xyz", "domain": "rgbd", "modes": ["allocate", "reuse"] }, { "id": "rgbd_to_point_cloud", "domain": "spatial-e2e", "modes": ["allocate"] }, { "id": "ai_preprocess", "domain": "dnn-adapter", "modes": ["allocate", "reuse"] }, diff --git a/bench/opencv_comparison/run.py b/bench/opencv_comparison/run.py index 9db9d24..2e73070 100644 --- a/bench/opencv_comparison/run.py +++ b/bench/opencv_comparison/run.py @@ -21,6 +21,7 @@ "rgbd": ROOT / "bench" / "opencv_rgbd_comparison" / "run.py", "video": ROOT / "bench" / "opencv_video_comparison" / "run.py", "odometry": ROOT / "bench" / "opencv_odometry_comparison" / "run.py", + "photography": ROOT / "bench" / "opencv_photography_comparison" / "run.py", } diff --git a/bench/opencv_photography_comparison/README.md b/bench/opencv_photography_comparison/README.md new file mode 100644 index 0000000..3807de9 --- /dev/null +++ b/bench/opencv_photography_comparison/README.md @@ -0,0 +1,4 @@ +# OpenCV photography comparison + +Checks bounded panorama canvas geometry and source-only warp pixels against +OpenCV `warpPerspective`, plus deterministic gray-world channel balance. diff --git a/bench/opencv_photography_comparison/run.py b/bench/opencv_photography_comparison/run.py new file mode 100644 index 0000000..378cf0b --- /dev/null +++ b/bench/opencv_photography_comparison/run.py @@ -0,0 +1,61 @@ +"""OpenCV correctness receipt for Epic 108 panorama composition.""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +import cv2 +import numpy as np +import spatialrust as sr + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from opencv_comparison.report import emit_report, environment, make_report + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--output", type=Path) + args = parser.parse_args() + height, width = 32, 64 + x = np.arange(width, dtype=np.uint8)[None, :] + source = np.zeros((height, width, 3), dtype=np.uint8) + source[..., 0] = x + 80 + source[..., 1] = 20 + target = np.zeros_like(source) + target[..., 1] = 40 + target[..., 2] = 160 + homography = np.array([[1.0, 0.0, 24.0], [0.0, 1.0, 0.0], [0.0, 0.0, 1.0]]) + panorama, origin_x, origin_y = sr.stitch_panorama_pair(source, target, homography) + cv_warp = cv2.warpPerspective(source, homography, (width + 24, height)) + source_only_error = int(np.max(np.abs( + panorama[:, width:, :].astype(np.int16) - cv_warp[:, width:, :].astype(np.int16) + ))) + + cast = np.zeros((2, 2, 3), dtype=np.uint8) + cast[...] = [40, 80, 120] + balanced = sr.gray_world_white_balance_image(cast) + channel_means = balanced.mean(axis=(0, 1)) + balance_spread = float(channel_means.max() - channel_means.min()) + expected_shape = [height, width + 24, 3] + status = "pass" if ( + list(panorama.shape) == expected_shape and origin_x == 0 and origin_y == 0 + and source_only_error == 0 and balance_spread <= 1.0 + ) else "fail" + emit_report(make_report( + suite="opencv-photography", kind="correctness", status=status, + environment_receipt=environment( + opencv_version=cv2.__version__, spatialrust_version=sr.__version__), + results={ + "panorama_shape": list(panorama.shape), + "origin": [origin_x, origin_y], + "opencv_source_only_max_error": source_only_error, + "balanced_channel_mean_spread": balance_spread, + "thresholds": {"warp_max_error": 0, "channel_mean_spread": 1.0}, + }, + ), args.output) + + +if __name__ == "__main__": + main() diff --git a/crates/spatialrust-platform/src/stability.rs b/crates/spatialrust-platform/src/stability.rs index 4309e94..80b05f4 100644 --- a/crates/spatialrust-platform/src/stability.rs +++ b/crates/spatialrust-platform/src/stability.rs @@ -120,6 +120,7 @@ impl StabilityRegistry { "spatialrust-vision::optical-flow", "spatialrust-vision::video", "spatialrust-vision::odometry", + "spatialrust-vision::photography", "spatialrust-vision::ai-adapters", "spatialrust-gpu::GpuImage", ]; diff --git a/crates/spatialrust-py/spatialrust.pyi b/crates/spatialrust-py/spatialrust.pyi index b5ed469..71ccffd 100644 --- a/crates/spatialrust-py/spatialrust.pyi +++ b/crates/spatialrust-py/spatialrust.pyi @@ -28,7 +28,8 @@ __all__: list[str] = [ "apply_transform", "recenter", "scale", "normalize_unit_sphere", "merge", "centroid", "bounding_box", "oriented_bounding_box", "voxelize", "range_image", "rgbd_to_point_cloud", "depth_to_xyz", "calibrate_pinhole_camera", - "calibrate_fisheye_angles", "dense_flow_image", "filter2d_image", "gaussian_blur_image", + "calibrate_fisheye_angles", "dense_flow_image", "gray_world_white_balance_image", + "stitch_panorama_pair", "filter2d_image", "gaussian_blur_image", "median_blur_image", "bilateral_filter_image", "sobel_image", "scharr_image", "laplacian_image", "pyr_down_image", "pyr_up_image", "morphology_image", "threshold_image", "otsu_threshold_image", "adaptive_threshold_image", @@ -705,6 +706,13 @@ def estimate_rgbd_odometry( depth_scale: float = ..., threshold: float = ..., ) -> tuple[_F64Array, _F64Array, _BoolArray, int]: ... +def gray_world_white_balance_image(image: _U8Array) -> _U8Array: ... +def stitch_panorama_pair( + source: _U8Array, + target: _U8Array, + homography: _F64Array, + max_output_pixels: int = ..., +) -> tuple[_U8Array, int, int]: ... def stereo_block_match( left: _U8Array, right: _U8Array, diff --git a/crates/spatialrust-py/src/lib.rs b/crates/spatialrust-py/src/lib.rs index 67675e9..5e0c277 100644 --- a/crates/spatialrust-py/src/lib.rs +++ b/crates/spatialrust-py/src/lib.rs @@ -79,8 +79,8 @@ use spatialrust::vision::{ estimate_homography_ransac as estimate_homography_ransac_op, estimate_rgbd_odometry as estimate_rgbd_odometry_op, filter2d as filter2d_op, find_contours as trace_contours, gaussian_blur as gaussian_blur_op, - histogram_u8 as histogram_u8_op, integral_image as integral_image_op, - laplacian as laplacian_op, letterbox as letterbox_op, + gray_world_white_balance as gray_world_white_balance_op, histogram_u8 as histogram_u8_op, + integral_image as integral_image_op, laplacian as laplacian_op, letterbox as letterbox_op, match_descriptors as match_descriptors_op, median_blur as median_blur_op, morphology_ex as morphology_ex_op, nms as nms_op, otsu_threshold_u8 as otsu_threshold_u8_op, pack_chw as pack_chw_op, pack_chw_into as pack_chw_into_op, @@ -89,11 +89,12 @@ use spatialrust::vision::{ rgb_to_gray as rgb_to_gray_op, rgb_to_gray_into as rgb_to_gray_into_op, rgb_to_hsv as rgb_to_hsv_op, scharr as scharr_op, sobel as sobel_op, soft_nms as soft_nms_op, solve_pnp as solve_pnp_op, stereo_block_match as stereo_block_match_op, - threshold as threshold_op, AbsolutePose, AdaptiveThresholdMethod, BinaryMask, BorderMode, - BoundingBox2, CameraMatrix3, CannyOptions, ConfidenceMap, Connectivity, CornerSelectionOptions, - DescriptorBuffer, FastOptions, HarrisOptions, Interpolation, Kernel2D, Keypoint2, MaskRle, - MatchOptions, MorphologyOperation, MorphologyShape, ObjectImageCorrespondence, OrbOptions, - OrbScoreType, PointCorrespondence2, PointMap, RgbdOdometryOptions, RleOrder, + stitch_panorama_pair as stitch_panorama_pair_op, threshold as threshold_op, AbsolutePose, + AdaptiveThresholdMethod, BinaryMask, BorderMode, BoundingBox2, CameraMatrix3, CannyOptions, + ConfidenceMap, Connectivity, CornerSelectionOptions, DescriptorBuffer, FastOptions, + HarrisOptions, Interpolation, Kernel2D, Keypoint2, MaskRle, MatchOptions, MorphologyOperation, + MorphologyShape, ObjectImageCorrespondence, OrbOptions, OrbScoreType, PanoramaOptions, + PerspectiveTransform, PointCorrespondence2, PointMap, RgbdOdometryOptions, RleOrder, RobustEstimationOptions, ShiTomasiOptions, SoftNmsMethod, StereoBmOptions, StructuringElement, ThresholdType, }; @@ -789,6 +790,56 @@ fn dense_flow_image<'py>( Ok(array.into_pyarray_bound(py)) } +/// Applies gray-world white balance to an RGB image. +#[pyfunction] +fn gray_world_white_balance_image<'py>( + py: Python<'py>, + image: PyReadonlyArray3<'_, u8>, +) -> PyResult>> { + let image = rgb_image_from_numpy(image)?; + let output = gray_world_white_balance_op(image.view()).map_err(to_py_err)?; + let array = Array3::from_shape_vec((output.height(), output.width(), 3), output.into_vec()) + .map_err(to_py_err)?; + Ok(array.into_pyarray_bound(py)) +} + +/// Stitches a source RGB image into target coordinates with a 3x3 homography. +#[pyfunction] +#[pyo3(signature = (source, target, homography, max_output_pixels=67108864))] +fn stitch_panorama_pair<'py>( + py: Python<'py>, + source: PyReadonlyArray3<'_, u8>, + target: PyReadonlyArray3<'_, u8>, + homography: PyReadonlyArray2<'_, f64>, + max_output_pixels: usize, +) -> PyResult<(Bound<'py, PyArray3>, i32, i32)> { + let source = rgb_image_from_numpy(source)?; + let target = rgb_image_from_numpy(target)?; + let matrix = homography.as_array(); + if matrix.shape() != [3, 3] { + return Err(PyValueError::new_err("homography must have shape (3, 3)")); + } + let panorama = stitch_panorama_pair_op( + source.view(), + target.view(), + PerspectiveTransform { + matrix: [ + [matrix[[0, 0]], matrix[[0, 1]], matrix[[0, 2]]], + [matrix[[1, 0]], matrix[[1, 1]], matrix[[1, 2]]], + [matrix[[2, 0]], matrix[[2, 1]], matrix[[2, 2]]], + ], + }, + PanoramaOptions { max_output_pixels }, + ) + .map_err(to_py_err)?; + let (origin_x, origin_y) = (panorama.origin_x(), panorama.origin_y()); + let image = panorama.image(); + let array = + Array3::from_shape_vec((image.height(), image.width(), 3), image.as_slice().to_vec()) + .map_err(to_py_err)?; + Ok((array.into_pyarray_bound(py), origin_x, origin_y)) +} + fn cloud_from_xyz(arr: PyReadonlyArray2<'_, f32>) -> PyResult { let view = arr.as_array(); let shape = view.shape(); @@ -3188,6 +3239,8 @@ fn spatialrust_module(m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_function(wrap_pyfunction!(estimate_homography_ransac, m)?)?; m.add_function(wrap_pyfunction!(solve_pnp, m)?)?; m.add_function(wrap_pyfunction!(estimate_rgbd_odometry, m)?)?; + m.add_function(wrap_pyfunction!(gray_world_white_balance_image, m)?)?; + m.add_function(wrap_pyfunction!(stitch_panorama_pair, m)?)?; m.add_function(wrap_pyfunction!(stereo_block_match, m)?)?; m.add_function(wrap_pyfunction!(match_binary_descriptors, m)?)?; m.add_function(wrap_pyfunction!(match_float_descriptors, m)?)?; diff --git a/crates/spatialrust-vision/Cargo.toml b/crates/spatialrust-vision/Cargo.toml index 06c997f..13c132f 100644 --- a/crates/spatialrust-vision/Cargo.toml +++ b/crates/spatialrust-vision/Cargo.toml @@ -20,13 +20,14 @@ imgproc-canny = ["imgproc-filter"] feature2d = ["imgproc-filter", "resize"] geometry = ["dep:spatialrust-camera"] odometry = ["geometry", "feature2d"] +photography = ["warp", "geometry"] detection = [] dense = ["detection"] spatial = ["dense", "dep:spatialrust-core", "dep:spatialrust-camera"] ai-adapters = ["preprocess", "dense", "detection", "dep:spatialrust-tensor", "dep:bytemuck", "spatialrust-tensor/image"] video = ["dense", "detection"] video-adapters = ["video"] -full = ["preprocess", "warp", "imgproc-filter", "imgproc-morphology", "imgproc-analysis", "imgproc-canny", "feature2d", "geometry", "odometry", "detection", "dense", "spatial", "ai-adapters", "video-adapters"] +full = ["preprocess", "warp", "imgproc-filter", "imgproc-morphology", "imgproc-analysis", "imgproc-canny", "feature2d", "geometry", "odometry", "photography", "detection", "dense", "spatial", "ai-adapters", "video-adapters"] [dependencies] spatialrust-image.workspace = true @@ -90,3 +91,8 @@ required-features = ["video"] name = "odometry" harness = false required-features = ["odometry"] + +[[bench]] +name = "photography" +harness = false +required-features = ["photography"] diff --git a/crates/spatialrust-vision/benches/photography.rs b/crates/spatialrust-vision/benches/photography.rs new file mode 100644 index 0000000..611eed6 --- /dev/null +++ b/crates/spatialrust-vision/benches/photography.rs @@ -0,0 +1,21 @@ +use criterion::{black_box, criterion_group, criterion_main, Criterion}; +use spatialrust_image::Image; +use spatialrust_vision::{fuse_exposures, ExposureFusionOptions}; + +fn benchmark_photography(c: &mut Criterion) { + let images = [48_u8, 128, 220].map(|value| Image::from_pixel(640, 480, [value; 3]).unwrap()); + c.bench_function("exposure_fusion_vga_3", |b| { + b.iter(|| { + black_box( + fuse_exposures( + &[images[0].view(), images[1].view(), images[2].view()], + ExposureFusionOptions::default(), + ) + .unwrap(), + ) + }) + }); +} + +criterion_group!(benches, benchmark_photography); +criterion_main!(benches); diff --git a/crates/spatialrust-vision/src/lib.rs b/crates/spatialrust-vision/src/lib.rs index 9281eea..40a15c6 100644 --- a/crates/spatialrust-vision/src/lib.rs +++ b/crates/spatialrust-vision/src/lib.rs @@ -39,6 +39,8 @@ mod multiview; mod optical_flow; #[cfg(feature = "odometry")] mod odometry; +#[cfg(feature = "photography")] +mod photography; #[cfg(feature = "geometry")] mod pnp; #[cfg(feature = "ai-adapters")] @@ -91,6 +93,8 @@ pub use multiview::*; pub use optical_flow::*; #[cfg(feature = "odometry")] pub use odometry::*; +#[cfg(feature = "photography")] +pub use photography::*; #[cfg(feature = "geometry")] pub use pnp::*; #[cfg(feature = "feature2d")] diff --git a/crates/spatialrust-vision/src/photography.rs b/crates/spatialrust-vision/src/photography.rs new file mode 100644 index 0000000..d6e3e76 --- /dev/null +++ b/crates/spatialrust-vision/src/photography.rs @@ -0,0 +1,303 @@ +//! Runtime-free computational photography and pairwise panorama composition. + +use spatialrust_image::{Image, ImageView}; + +use crate::{ + estimate_homography_ransac, PerspectiveTransform, PointCorrespondence2, + RobustEstimationOptions, VisionError, VisionResult, +}; + +/// Applies deterministic gray-world white balance to an RGB image. +pub fn gray_world_white_balance(input: ImageView<'_, u8, 3>) -> VisionResult> { + if input.width() == 0 || input.height() == 0 { + return Ok(Image::try_new_with_metadata(0, 0, Vec::new(), input.metadata())?); + } + let pixels = input.width().saturating_mul(input.height()) as f64; + let mut sums = [0.0; 3]; + for y in 0..input.height() { + let row = input.row(y).expect("validated RGB row"); + for pixel in row.chunks_exact(3) { + for channel in 0..3 { + sums[channel] += f64::from(pixel[channel]); + } + } + } + let means = sums.map(|sum| sum / pixels); + let target = means.iter().sum::() / 3.0; + let gains = means.map(|mean| if mean > 0.0 { target / mean } else { 1.0 }); + let mut data = Vec::with_capacity(input.width() * input.height() * 3); + for y in 0..input.height() { + for (index, &value) in input.row(y).expect("validated RGB row").iter().enumerate() { + data.push((f64::from(value) * gains[index % 3]).round().clamp(0.0, 255.0) as u8); + } + } + Ok(Image::try_new_with_metadata(input.width(), input.height(), data, input.metadata())?) +} + +/// Well-exposedness fusion controls. +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct ExposureFusionOptions { + /// Standard deviation around middle gray in normalized intensity. + pub well_exposed_sigma: f64, + /// Positive floor preventing completely black weights. + pub weight_floor: f64, +} + +impl Default for ExposureFusionOptions { + fn default() -> Self { + Self { well_exposed_sigma: 0.2, weight_floor: 1e-6 } + } +} + +/// Fuses aligned RGB exposures with normalized per-pixel well-exposedness. +pub fn fuse_exposures( + inputs: &[ImageView<'_, u8, 3>], + options: ExposureFusionOptions, +) -> VisionResult> { + let first = *inputs.first().ok_or_else(|| { + VisionError::InvalidParameter("exposure fusion requires at least one image".into()) + })?; + if !options.well_exposed_sigma.is_finite() + || options.well_exposed_sigma <= 0.0 + || !options.weight_floor.is_finite() + || options.weight_floor <= 0.0 + { + return Err(VisionError::InvalidParameter("invalid exposure fusion weights".into())); + } + if inputs.iter().any(|image| { + image.width() != first.width() + || image.height() != first.height() + || image.metadata() != first.metadata() + }) { + return Err(VisionError::ShapeMismatch( + "exposure inputs must share dimensions and metadata".into(), + )); + } + let mut output = Vec::with_capacity(first.width() * first.height() * 3); + let sigma2 = 2.0 * options.well_exposed_sigma.powi(2); + for y in 0..first.height() { + for x in 0..first.width() { + let mut weighted = [0.0; 3]; + let mut total = 0.0; + for image in inputs { + let pixel = image.get(x, y).expect("validated coordinates"); + let luminance = (0.2126 * f64::from(pixel[0]) + + 0.7152 * f64::from(pixel[1]) + + 0.0722 * f64::from(pixel[2])) + / 255.0; + let weight = (-(luminance - 0.5).powi(2) / sigma2).exp() + options.weight_floor; + for channel in 0..3 { + weighted[channel] += f64::from(pixel[channel]) * weight; + } + total += weight; + } + output.extend(weighted.map(|value| (value / total).round().clamp(0.0, 255.0) as u8)); + } + } + Ok(Image::try_new_with_metadata(first.width(), first.height(), output, first.metadata())?) +} + +/// Bounded pairwise panorama settings. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PanoramaOptions { + /// Hard ceiling for output pixels before allocation. + pub max_output_pixels: usize, +} + +impl Default for PanoramaOptions { + fn default() -> Self { + Self { max_output_pixels: 64 * 1024 * 1024 } + } +} + +/// Pairwise panorama and its integer world-coordinate origin. +#[derive(Clone, Debug, PartialEq)] +pub struct Panorama { + image: Image, + origin_x: i32, + origin_y: i32, +} + +impl Panorama { + /// Returns the blended RGB canvas. + pub const fn image(&self) -> &Image { + &self.image + } + /// Returns the right-image world x coordinate represented by canvas x=0. + pub const fn origin_x(&self) -> i32 { + self.origin_x + } + /// Returns the right-image world y coordinate represented by canvas y=0. + pub const fn origin_y(&self) -> i32 { + self.origin_y + } +} + +/// Estimates a source-to-target homography and stitches the pair. +pub fn estimate_and_stitch_panorama( + source: ImageView<'_, u8, 3>, + target: ImageView<'_, u8, 3>, + correspondences: &[PointCorrespondence2], + robust: RobustEstimationOptions, + options: PanoramaOptions, +) -> VisionResult { + let estimate = estimate_homography_ransac(correspondences, robust)?; + stitch_panorama_pair( + source, + target, + PerspectiveTransform { matrix: estimate.model().matrix().m }, + options, + ) +} + +/// Warps `source` into `target` coordinates and feather-blends their overlap. +pub fn stitch_panorama_pair( + source: ImageView<'_, u8, 3>, + target: ImageView<'_, u8, 3>, + source_to_target: PerspectiveTransform, + options: PanoramaOptions, +) -> VisionResult { + if source.metadata() != target.metadata() { + return Err(VisionError::ShapeMismatch("panorama images must share color metadata".into())); + } + if source.width() == 0 || source.height() == 0 || target.width() == 0 || target.height() == 0 { + return Err(VisionError::InvalidDimensions("panorama inputs must be non-empty".into())); + } + if options.max_output_pixels == 0 { + return Err(VisionError::InvalidParameter("panorama pixel budget must be positive".into())); + } + let mut corners = vec![ + (0.0, 0.0), + ((target.width() - 1) as f64, 0.0), + (0.0, (target.height() - 1) as f64), + ((target.width() - 1) as f64, (target.height() - 1) as f64), + ]; + for &(x, y) in &[ + (0.0, 0.0), + ((source.width() - 1) as f64, 0.0), + (0.0, (source.height() - 1) as f64), + ((source.width() - 1) as f64, (source.height() - 1) as f64), + ] { + corners.push(source_to_target.map_point(x, y).ok_or(VisionError::SingularTransform)?); + } + let min_x = corners.iter().map(|p| p.0).fold(f64::INFINITY, f64::min).floor() as i32; + let min_y = corners.iter().map(|p| p.1).fold(f64::INFINITY, f64::min).floor() as i32; + let max_x = corners.iter().map(|p| p.0).fold(f64::NEG_INFINITY, f64::max).ceil() as i32; + let max_y = corners.iter().map(|p| p.1).fold(f64::NEG_INFINITY, f64::max).ceil() as i32; + let width = usize::try_from(max_x - min_x + 1).map_err(|_| { + VisionError::InvalidDimensions("panorama width is not representable".into()) + })?; + let height = usize::try_from(max_y - min_y + 1).map_err(|_| { + VisionError::InvalidDimensions("panorama height is not representable".into()) + })?; + if width.checked_mul(height).filter(|&pixels| pixels <= options.max_output_pixels).is_none() { + return Err(VisionError::InvalidDimensions("panorama exceeds output pixel budget".into())); + } + let inverse = source_to_target.inverse()?; + let mut data = Vec::with_capacity(width * height * 3); + for canvas_y in 0..height { + for canvas_x in 0..width { + let world_x = f64::from(min_x) + canvas_x as f64; + let world_y = f64::from(min_y) + canvas_y as f64; + let target_sample = sample_inside(target, world_x, world_y); + let source_sample = + inverse.map_point(world_x, world_y).and_then(|(x, y)| sample_inside(source, x, y)); + let pixel = match (source_sample, target_sample) { + (Some((left, lw)), Some((right, rw))) => { + let total = lw + rw; + std::array::from_fn(|channel| { + ((f64::from(left[channel]) * lw + f64::from(right[channel]) * rw) / total) + .round() + .clamp(0.0, 255.0) as u8 + }) + } + (Some((pixel, _)), None) | (None, Some((pixel, _))) => pixel, + (None, None) => [0; 3], + }; + data.extend_from_slice(&pixel); + } + } + Ok(Panorama { + image: Image::try_new_with_metadata(width, height, data, target.metadata())?, + origin_x: min_x, + origin_y: min_y, + }) +} + +fn sample_inside(image: ImageView<'_, u8, 3>, x: f64, y: f64) -> Option<([u8; 3], f64)> { + if x < 0.0 || y < 0.0 || x > (image.width() - 1) as f64 || y > (image.height() - 1) as f64 { + return None; + } + let x0 = x.floor() as usize; + let y0 = y.floor() as usize; + let x1 = (x0 + 1).min(image.width() - 1); + let y1 = (y0 + 1).min(image.height() - 1); + let wx = x - x0 as f64; + let wy = y - y0 as f64; + let pixel = std::array::from_fn(|channel| { + let p00 = f64::from(image.get(x0, y0).unwrap()[channel]); + let p10 = f64::from(image.get(x1, y0).unwrap()[channel]); + let p01 = f64::from(image.get(x0, y1).unwrap()[channel]); + let p11 = f64::from(image.get(x1, y1).unwrap()[channel]); + let top = p00 * (1.0 - wx) + p10 * wx; + let bottom = p01 * (1.0 - wx) + p11 * wx; + (top * (1.0 - wy) + bottom * wy).round().clamp(0.0, 255.0) as u8 + }); + let edge = + x.min(y).min((image.width() - 1) as f64 - x).min((image.height() - 1) as f64 - y) + 1.0; + Some((pixel, edge.max(1e-6))) +} + +#[cfg(test)] +mod tests { + use super::{ + fuse_exposures, gray_world_white_balance, stitch_panorama_pair, ExposureFusionOptions, + PanoramaOptions, + }; + use crate::PerspectiveTransform; + use spatialrust_image::Image; + + #[test] + fn white_balance_equalizes_channel_means() { + let image = Image::try_new(2, 1, vec![40, 80, 120, 20, 40, 60]).unwrap(); + let balanced = gray_world_white_balance(image.view()).unwrap(); + let sums = [ + balanced.as_slice()[0] as u16 + balanced.as_slice()[3] as u16, + balanced.as_slice()[1] as u16 + balanced.as_slice()[4] as u16, + balanced.as_slice()[2] as u16 + balanced.as_slice()[5] as u16, + ]; + assert!(sums.iter().max().unwrap() - sums.iter().min().unwrap() <= 1); + } + + #[test] + fn exposure_fusion_prefers_middle_gray() { + let dark = Image::from_pixel(1, 1, [10, 10, 10]).unwrap(); + let middle = Image::from_pixel(1, 1, [128, 128, 128]).unwrap(); + let fused = fuse_exposures(&[dark.view(), middle.view()], ExposureFusionOptions::default()) + .unwrap(); + assert!(fused.as_slice()[0] >= 120); + } + + #[test] + fn translated_pair_expands_canvas_and_preserves_sides() { + let source = Image::from_pixel(3, 2, [200, 0, 0]).unwrap(); + let target = Image::from_pixel(3, 2, [0, 0, 200]).unwrap(); + let panorama = stitch_panorama_pair( + source.view(), + target.view(), + PerspectiveTransform { matrix: [[1.0, 0.0, 2.0], [0.0, 1.0, 0.0], [0.0, 0.0, 1.0]] }, + PanoramaOptions::default(), + ) + .unwrap(); + assert_eq!((panorama.image().width(), panorama.image().height()), (5, 2)); + assert_eq!(panorama.image().get(0, 0).unwrap(), &[0, 0, 200]); + assert_eq!(panorama.image().get(4, 0).unwrap(), &[200, 0, 0]); + assert!(stitch_panorama_pair( + source.view(), + target.view(), + PerspectiveTransform::identity(), + PanoramaOptions { max_output_pixels: 5 }, + ) + .is_err()); + } +} diff --git a/crates/spatialrust/Cargo.toml b/crates/spatialrust/Cargo.toml index 629e236..e990897 100644 --- a/crates/spatialrust/Cargo.toml +++ b/crates/spatialrust/Cargo.toml @@ -155,6 +155,7 @@ vision-imgproc-canny = ["vision-imgproc-filter", "spatialrust-vision/imgproc-can vision-feature2d = ["vision", "spatialrust-vision/feature2d"] vision-geometry = ["vision", "camera", "spatialrust-vision/geometry"] vision-odometry = ["vision-geometry", "vision-feature2d", "spatialrust-vision/odometry"] +vision-photography = ["vision-warp", "vision-geometry", "spatialrust-vision/photography"] mapping-vision-odometry = ["vision-odometry", "mapping", "spatialrust-mapping/vision-odometry"] vision-detection = ["vision", "spatialrust-vision/detection"] vision-dense = ["vision-detection", "spatialrust-vision/dense"] @@ -172,6 +173,7 @@ vision-full = [ "vision-feature2d", "vision-geometry", "vision-odometry", + "vision-photography", "vision-detection", "vision-dense", "vision-spatial", diff --git a/docs/API_STABILITY.md b/docs/API_STABILITY.md index 26fa27e..592bfe1 100644 --- a/docs/API_STABILITY.md +++ b/docs/API_STABILITY.md @@ -63,7 +63,7 @@ until their individual 1.0 milestones. | Camera (`camera`, `camera-rgbd`) | Pinhole/Brown-Conrady models and explicit RGB-D conversion entry points are stable | | Image IO (`image-io-*`) | Bounded codecs, typed decoded pixels, and source metadata are provisional | | AI (`ai-*`) | Backend/session, named dynamic I/O, copy policy, I/O binding, mock backend, and ONNX Runtime adapter APIs are provisional | -| Vision (`vision-*`) | Base errors/borders, resize/filter entry points, detection/dense data contracts, and Feature2D data contracts are stable; geometry, stereo, optical flow, odometry, video, and AI adapters remain provisional | +| Vision (`vision-*`) | Base errors/borders, resize/filter entry points, detection/dense data contracts, and Feature2D data contracts are stable; geometry, stereo, optical flow, odometry, photography, video, and AI adapters remain provisional | | Tensor (`tensor-*`) | Dtype/layout/device ownership, typed host storage, external host owner, and DLPack APIs are provisional | | Records (`records`) | Versioned `SpatialRecord`, schema compatibility/migration, and chunked record streams are provisional | | Arrow (`arrow-*`) | Arrow C Data/Stream/Device bridges for point clouds are provisional | diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index ac382ac..7862e46 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -87,6 +87,9 @@ Visual and RGB-D odometry kernels remain in the additive `odometry` vision feature. Their conversion into stamped trajectory motion is a one-way optional bridge in `spatialrust-mapping`; monocular scale and invalid depth remain explicit at that boundary. +Computational photography composes image/warp/geometry primitives behind the +additive `photography` feature. Panorama APIs expose canvas bounds and enforce a +pre-allocation pixel budget; codec and device execution remain separate. Its `imgproc-*` features share one border extrapolation contract; `filter2d` means correlation, while true convolution is an explicitly named operation. `spatialrust-tensor` is distinct from the point-cloud chunk iterator named diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index bfc054a..69795cf 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -384,7 +384,7 @@ uses the standard completion gates above and lands as one reviewable PR. | 105 | Complete | 88, 101–102 | Mono/stereo/fisheye/hand-eye calibration and bundle-adjustment contracts | | 106 | Complete | 92, 101–105 | Dense flow, tracking, background modeling, and feature-gated video stream adapters | | 107 | Complete | 93, 101–106 | Stronger local features, robust tracking, and visual/RGB-D odometry integration | -| 108 | Planned | 101–107 | Feature-gated computational photography and panorama stitching | +| 108 | Complete | 101–107 | Feature-gated computational photography and panorama stitching | | 109 | Planned | 97, 99, 101–108 | Bounded spatial execution graph with fusion, backpressure, and named transfer receipts | | 110 | Planned | 100, 101–109 | SpatialRust Vision 1.0 conformance, audits, performance budgets, examples, and migration policy | @@ -500,3 +500,18 @@ The vision layer reports source-to-target motion and never invents monocular scale. `spatialrust-mapping` accepts scale explicitly for monocular estimates and preserves metric RGB-D translation. Invalid source depths are counted, not silently filled or copied to another device. + +### Epic 108 delivery slices + +| Slice | Status | Scope | Evidence | +| --- | --- | --- | --- | +| 108A | Complete | Deterministic RGB gray-world white balance | channel-mean equality test and Python binding | +| 108B | Complete | Aligned well-exposedness fusion | middle-gray preference test and VGA/3-exposure Criterion | +| 108C | Complete | Bounded pairwise panorama canvas and origin receipt | translated pair geometry and pixel-budget rejection | +| 108D | Complete | Bilinear source warp and edge-distance feather blending | overlap/non-overlap known-pixel tests | +| 108E | Complete | Homography estimation composition and OpenCV comparison | RANSAC entry point and zero-error `warpPerspective` receipt | + +Photography remains a runtime-free `vision-photography` feature. Inputs must +share dimensions/metadata where alignment requires it, panorama allocations are +checked against a caller-visible pixel ceiling, and no codec or GPU transfer is +performed implicitly. diff --git a/notes/2026-07-15_epic108_photography.md b/notes/2026-07-15_epic108_photography.md new file mode 100644 index 0000000..165cdd5 --- /dev/null +++ b/notes/2026-07-15_epic108_photography.md @@ -0,0 +1,11 @@ +# Epic 108: computational photography and panorama + +Epic 108 adds runtime-free gray-world correction, aligned exposure fusion, and +bounded pairwise panorama composition. Source-to-target transforms and output +origins are explicit, image metadata must agree, and the output pixel ceiling +is checked before allocation. + +The OpenCV 4.10 receipt produced a `32x88` panorama at origin `(0, 0)`. +SpatialRust's source-only warped region matched `cv2.warpPerspective` exactly +(maximum component error `0`), and gray-world correction reduced the synthetic +channel-mean spread to `0.0`.