diff --git a/CHANGELOG.md b/CHANGELOG.md index 451ddb0..d950d96 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,6 +21,12 @@ removed no sooner than the next major (see `docs/API_STABILITY.md`). ### Added +- **Video recognition substrate (Epic 106)**: dense integer block flow, + adaptive Gaussian foreground segmentation, deterministic same-class IoU + tracking, and timestamped pull-based video source contracts. Codec/camera + adapters remain feature-gated; Python dense flow, OpenCV Farneback comparison, + and QQVGA/QVGA Criterion workloads cover the portable core. + - **Camera calibration contracts (Epic 105)**: robust pinhole intrinsics, Kannala–Brandt4 fisheye fitting, supplied-rotation stereo and hand-eye translation solves, and fixed-camera sparse point bundle adjustment. All diff --git a/bench/opencv_comparison/manifest.json b/bench/opencv_comparison/manifest.json index 6c7f7d5..ef44324 100644 --- a/bench/opencv_comparison/manifest.json +++ b/bench/opencv_comparison/manifest.json @@ -25,6 +25,9 @@ { "id": "stereo_bm", "domain": "calib3d", "modes": ["allocate"] }, { "id": "pinhole_calibration", "domain": "calib3d", "modes": ["allocate"] }, { "id": "fisheye_calibration", "domain": "calib3d", "modes": ["allocate"] }, + { "id": "dense_optical_flow", "domain": "video", "modes": ["allocate"] }, + { "id": "background_model", "domain": "video", "modes": ["streaming"] }, + { "id": "multi_object_tracking", "domain": "video", "modes": ["streaming"] }, { "id": "depth_to_xyz", "domain": "rgbd", "modes": ["allocate", "reuse"] }, { "id": "rgbd_to_point_cloud", "domain": "spatial-e2e", "modes": ["allocate"] }, { "id": "ai_preprocess", "domain": "dnn-adapter", "modes": ["allocate", "reuse"] }, diff --git a/bench/opencv_comparison/run.py b/bench/opencv_comparison/run.py index 482aa0a..7ebe11c 100644 --- a/bench/opencv_comparison/run.py +++ b/bench/opencv_comparison/run.py @@ -19,6 +19,7 @@ / "opencv_vision_comparison" / "performance.py", "rgbd": ROOT / "bench" / "opencv_rgbd_comparison" / "run.py", + "video": ROOT / "bench" / "opencv_video_comparison" / "run.py", } diff --git a/bench/opencv_video_comparison/README.md b/bench/opencv_video_comparison/README.md new file mode 100644 index 0000000..b06cf17 --- /dev/null +++ b/bench/opencv_video_comparison/README.md @@ -0,0 +1,9 @@ +# OpenCV video comparison + +This suite compares SpatialRust dense integer block flow with OpenCV Farneback +flow on a deterministic translated texture: + +```bash +python bench/opencv_video_comparison/run.py \ + --output target/opencv-comparison/video.json +``` diff --git a/bench/opencv_video_comparison/run.py b/bench/opencv_video_comparison/run.py new file mode 100644 index 0000000..c132dcd --- /dev/null +++ b/bench/opencv_video_comparison/run.py @@ -0,0 +1,81 @@ +"""Dense-flow correctness comparison for Epic 106 video recognition.""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +import cv2 +import numpy as np +import spatialrust as sr + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from opencv_comparison.report import emit_report, environment, make_report + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("--output", type=Path) + return parser.parse_args() + + +def main() -> None: + args = parse_args() + height, width = 96, 128 + dx, dy = 3.0, 2.0 + yy, xx = np.indices((height, width), dtype=np.int32) + previous = ((xx * 17 + yy * 29 + xx * yy * 3) % 251).astype(np.uint8) + transform = np.array([[1.0, 0.0, dx], [0.0, 1.0, dy]], dtype=np.float32) + next_frame = cv2.warpAffine( + previous, + transform, + (width, height), + flags=cv2.INTER_NEAREST, + borderMode=cv2.BORDER_CONSTANT, + ) + spatialrust_flow = sr.dense_flow_image(previous, next_frame, search_radius=4) + opencv_flow = cv2.calcOpticalFlowFarneback( + previous, + next_frame, + None, + pyr_scale=0.5, + levels=3, + winsize=15, + iterations=5, + poly_n=5, + poly_sigma=1.1, + flags=0, + ) + region = np.s_[20:-20, 20:-20] + spatialrust_median = np.nanmedian(spatialrust_flow[region], axis=(0, 1)) + opencv_median = np.median(opencv_flow[region], axis=(0, 1)) + expected = np.array([dx, dy]) + spatialrust_error = float(np.max(np.abs(spatialrust_median - expected))) + opencv_error = float(np.max(np.abs(opencv_median - expected))) + cross_error = float(np.max(np.abs(spatialrust_median - opencv_median))) + status = "pass" if spatialrust_error <= 0.01 and cross_error <= 0.5 else "fail" + report = make_report( + suite="opencv-video", + kind="correctness", + status=status, + environment_receipt=environment( + opencv_version=cv2.__version__, spatialrust_version=sr.__version__ + ), + results={ + "translation": [dx, dy], + "spatialrust_median_flow": spatialrust_median.tolist(), + "opencv_median_flow": opencv_median.tolist(), + "spatialrust_max_error": spatialrust_error, + "opencv_max_error": opencv_error, + "cross_max_error": cross_error, + "thresholds": {"spatialrust_max_error": 0.01, "cross_max_error": 0.5}, + }, + ) + emit_report(report, args.output) + if status != "pass": + raise SystemExit(1) + + +if __name__ == "__main__": + main() diff --git a/crates/spatialrust-platform/src/stability.rs b/crates/spatialrust-platform/src/stability.rs index 4884dbf..6a2b8db 100644 --- a/crates/spatialrust-platform/src/stability.rs +++ b/crates/spatialrust-platform/src/stability.rs @@ -118,6 +118,7 @@ impl StabilityRegistry { "spatialrust-vision::geometry", "spatialrust-vision::stereo", "spatialrust-vision::optical-flow", + "spatialrust-vision::video", "spatialrust-vision::ai-adapters", "spatialrust-gpu::GpuImage", ]; diff --git a/crates/spatialrust-py/spatialrust.pyi b/crates/spatialrust-py/spatialrust.pyi index 1bcd632..2e9ad9a 100644 --- a/crates/spatialrust-py/spatialrust.pyi +++ b/crates/spatialrust-py/spatialrust.pyi @@ -28,7 +28,7 @@ __all__: list[str] = [ "apply_transform", "recenter", "scale", "normalize_unit_sphere", "merge", "centroid", "bounding_box", "oriented_bounding_box", "voxelize", "range_image", "rgbd_to_point_cloud", "depth_to_xyz", "calibrate_pinhole_camera", - "calibrate_fisheye_angles", "filter2d_image", "gaussian_blur_image", + "calibrate_fisheye_angles", "dense_flow_image", "filter2d_image", "gaussian_blur_image", "median_blur_image", "bilateral_filter_image", "sobel_image", "scharr_image", "laplacian_image", "pyr_down_image", "pyr_up_image", "morphology_image", "threshold_image", "otsu_threshold_image", "adaptive_threshold_image", @@ -129,6 +129,12 @@ def calibrate_pinhole_camera( def calibrate_fisheye_angles( theta: _F64Array, distorted_radius: _F64Array ) -> tuple[float, float, float, float, float]: ... +def dense_flow_image( + previous: _U8Array, + next: _U8Array, + block_radius: int = ..., + search_radius: int = ..., +) -> _F32Array: ... def rgbd_to_point_cloud( depth: _F32Array, diff --git a/crates/spatialrust-py/src/lib.rs b/crates/spatialrust-py/src/lib.rs index fa57255..3144a44 100644 --- a/crates/spatialrust-py/src/lib.rs +++ b/crates/spatialrust-py/src/lib.rs @@ -95,6 +95,7 @@ use spatialrust::vision::{ OrbScoreType, PointCorrespondence2, PointMap, RleOrder, RobustEstimationOptions, ShiTomasiOptions, SoftNmsMethod, StereoBmOptions, StructuringElement, ThresholdType, }; +use spatialrust::vision::{dense_flow_block_match as dense_flow_native, DenseFlowOptions}; use spatialrust::voxelize::{ range_image as range_image_proj, voxelize as voxelize_grid, RangeImageConfig, VoxelFill, VoxelGridConfig, @@ -760,6 +761,32 @@ fn calibrate_fisheye_angles( Ok((model.k1, model.k2, model.k3, model.k4, report.rms_residual)) } +/// Computes dense integer grayscale flow as an `(H, W, 2)` float32 array. +#[pyfunction] +#[pyo3(signature = (previous, next, block_radius=2, search_radius=4))] +fn dense_flow_image<'py>( + py: Python<'py>, + previous: PyReadonlyArray2<'_, u8>, + next: PyReadonlyArray2<'_, u8>, + block_radius: usize, + search_radius: usize, +) -> PyResult>> { + let previous = gray_u8_image_from_numpy(previous)?; + let next = gray_u8_image_from_numpy(next)?; + let flow = dense_flow_native( + previous.view(), + next.view(), + DenseFlowOptions { block_radius, search_radius, minimum_improvement: 1 }, + ) + .map_err(to_py_err)?; + let array = Array3::from_shape_vec( + (previous.height(), previous.width(), 2), + flow.image().as_slice().to_vec(), + ) + .map_err(to_py_err)?; + Ok(array.into_pyarray_bound(py)) +} + fn cloud_from_xyz(arr: PyReadonlyArray2<'_, f32>) -> PyResult { let view = arr.as_array(); let shape = view.shape(); @@ -3142,6 +3169,7 @@ fn spatialrust_module(m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_function(wrap_pyfunction!(rgbd_to_point_cloud, m)?)?; m.add_function(wrap_pyfunction!(calibrate_pinhole_camera, m)?)?; m.add_function(wrap_pyfunction!(calibrate_fisheye_angles, m)?)?; + m.add_function(wrap_pyfunction!(dense_flow_image, m)?)?; m.add_function(wrap_pyfunction!(filter2d_image, m)?)?; m.add_function(wrap_pyfunction!(gaussian_blur_image, m)?)?; m.add_function(wrap_pyfunction!(median_blur_image, m)?)?; diff --git a/crates/spatialrust-py/tests/test_bindings.py b/crates/spatialrust-py/tests/test_bindings.py index 155f51b..7f47234 100644 --- a/crates/spatialrust-py/tests/test_bindings.py +++ b/crates/spatialrust-py/tests/test_bindings.py @@ -771,3 +771,15 @@ def test_calibration_bindings_recover_intrinsics_and_fisheye(): fitted = sr.calibrate_fisheye_angles(theta, radius) np.testing.assert_allclose(fitted[:4], expected, atol=1e-9) assert fitted[4] < 1e-12 + + +def test_dense_flow_binding_recovers_translation_and_marks_border_invalid(): + height, width = 32, 40 + yy, xx = np.indices((height, width), dtype=np.int32) + previous = ((xx * 17 + yy * 29 + xx * yy * 3) % 251).astype(np.uint8) + next_frame = np.zeros_like(previous) + next_frame[1:, 2:] = previous[:-1, :-2] + flow = sr.dense_flow_image(previous, next_frame) + assert flow.shape == (height, width, 2) + np.testing.assert_array_equal(flow[16, 20], [2.0, 1.0]) + assert np.isnan(flow[0, 0]).all() diff --git a/crates/spatialrust-vision/Cargo.toml b/crates/spatialrust-vision/Cargo.toml index 789d041..6c9a2be 100644 --- a/crates/spatialrust-vision/Cargo.toml +++ b/crates/spatialrust-vision/Cargo.toml @@ -23,7 +23,9 @@ detection = [] dense = ["detection"] spatial = ["dense", "dep:spatialrust-core", "dep:spatialrust-camera"] ai-adapters = ["preprocess", "dense", "detection", "dep:spatialrust-tensor", "dep:bytemuck", "spatialrust-tensor/image"] -full = ["preprocess", "warp", "imgproc-filter", "imgproc-morphology", "imgproc-analysis", "imgproc-canny", "feature2d", "geometry", "detection", "dense", "spatial", "ai-adapters"] +video = ["dense", "detection"] +video-adapters = ["video"] +full = ["preprocess", "warp", "imgproc-filter", "imgproc-morphology", "imgproc-analysis", "imgproc-canny", "feature2d", "geometry", "detection", "dense", "spatial", "ai-adapters", "video-adapters"] [dependencies] spatialrust-image.workspace = true @@ -77,3 +79,8 @@ required-features = ["feature2d"] name = "geometry" harness = false required-features = ["geometry"] + +[[bench]] +name = "video" +harness = false +required-features = ["video"] diff --git a/crates/spatialrust-vision/benches/video.rs b/crates/spatialrust-vision/benches/video.rs new file mode 100644 index 0000000..6a36d8a --- /dev/null +++ b/crates/spatialrust-vision/benches/video.rs @@ -0,0 +1,32 @@ +use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; +use spatialrust_image::Image; +use spatialrust_vision::{dense_flow_block_match, DenseFlowOptions}; + +fn benchmark_video(c: &mut Criterion) { + for &(name, width, height) in &[("qqvga", 160, 120), ("qvga", 320, 240)] { + let previous = Image::::try_new( + width, + height, + (0..width * height).map(|index| ((index * 37) % 251) as u8).collect(), + ) + .unwrap(); + let next = previous.clone(); + let mut group = c.benchmark_group("dense_flow_block_match"); + group.sample_size(10); + group.throughput(Throughput::Elements((width * height) as u64)); + group.bench_function(BenchmarkId::from_parameter(name), |b| { + b.iter(|| { + dense_flow_block_match( + black_box(previous.view()), + black_box(next.view()), + DenseFlowOptions::default(), + ) + .unwrap() + }); + }); + group.finish(); + } +} + +criterion_group!(benches, benchmark_video); +criterion_main!(benches); diff --git a/crates/spatialrust-vision/src/lib.rs b/crates/spatialrust-vision/src/lib.rs index 3afec5a..f4ca603 100644 --- a/crates/spatialrust-vision/src/lib.rs +++ b/crates/spatialrust-vision/src/lib.rs @@ -53,6 +53,8 @@ mod spatial; mod stereo; #[cfg(feature = "warp")] mod warp; +#[cfg(feature = "video")] +mod video; pub use border::BorderMode; pub use error::{VisionError, VisionResult}; @@ -101,3 +103,5 @@ pub use spatial::*; pub use stereo::*; #[cfg(feature = "warp")] pub use warp::*; +#[cfg(feature = "video")] +pub use video::*; diff --git a/crates/spatialrust-vision/src/video.rs b/crates/spatialrust-vision/src/video.rs new file mode 100644 index 0000000..d3401d2 --- /dev/null +++ b/crates/spatialrust-vision/src/video.rs @@ -0,0 +1,507 @@ +//! Dense motion, background modeling, tracking, and video-source contracts. + +#[cfg(feature = "video-adapters")] +use std::collections::VecDeque; + +#[cfg(any(feature = "video-adapters", test))] +use spatialrust_image::Image; +use spatialrust_image::ImageView; + +use crate::{BinaryMask, BoundingBox2, Detection, FlowField, VisionError, VisionResult}; + +/// Integer block-matching controls for dense grayscale optical flow. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct DenseFlowOptions { + /// Half-size of the square comparison block. + pub block_radius: usize, + /// Maximum integer displacement searched in each axis. + pub search_radius: usize, + /// Minimum improvement over zero motion required for a valid vector. + pub minimum_improvement: u64, +} + +impl Default for DenseFlowOptions { + fn default() -> Self { + Self { block_radius: 2, search_radius: 4, minimum_improvement: 1 } + } +} + +/// Computes a dense integer flow field with deterministic local block matching. +/// +/// Pixels whose full block/search window leaves the image are marked with NaN. +pub fn dense_flow_block_match( + previous: ImageView<'_, u8, 1>, + next: ImageView<'_, u8, 1>, + options: DenseFlowOptions, +) -> VisionResult { + if (previous.width(), previous.height()) != (next.width(), next.height()) { + return Err(VisionError::ShapeMismatch( + "dense-flow frames must share dimensions".to_owned(), + )); + } + if options.block_radius == 0 || options.search_radius == 0 { + return Err(VisionError::InvalidParameter( + "dense-flow block/search radii must be positive".to_owned(), + )); + } + let margin = options.block_radius.saturating_add(options.search_radius); + if previous.width() <= margin * 2 || previous.height() <= margin * 2 { + return Err(VisionError::InvalidDimensions( + "dense-flow frames are too small for the configured search".to_owned(), + )); + } + let mut flow = vec![f32::NAN; previous.width() * previous.height() * 2]; + for y in margin..previous.height() - margin { + for x in margin..previous.width() - margin { + let zero_cost = block_cost(previous, next, x, y, 0, 0, options.block_radius); + let mut best_cost = zero_cost; + let mut best = (0_isize, 0_isize); + let search = options.search_radius as isize; + for dy in -search..=search { + for dx in -search..=search { + let cost = block_cost(previous, next, x, y, dx, dy, options.block_radius); + if cost < best_cost + || (cost == best_cost + && (dy.abs() + dx.abs(), dy, dx) + < (best.1.abs() + best.0.abs(), best.1, best.0)) + { + best_cost = cost; + best = (dx, dy); + } + } + } + if zero_cost.saturating_sub(best_cost) >= options.minimum_improvement { + let offset = (y * previous.width() + x) * 2; + flow[offset] = best.0 as f32; + flow[offset + 1] = best.1 as f32; + } + } + } + FlowField::try_new(previous.width(), previous.height(), flow) +} + +fn block_cost( + previous: ImageView<'_, u8, 1>, + next: ImageView<'_, u8, 1>, + x: usize, + y: usize, + dx: isize, + dy: isize, + radius: usize, +) -> u64 { + let mut cost = 0_u64; + let radius = radius as isize; + for by in -radius..=radius { + for bx in -radius..=radius { + let px = (x as isize + bx) as usize; + let py = (y as isize + by) as usize; + let nx = (x as isize + bx + dx) as usize; + let ny = (y as isize + by + dy) as usize; + let left = previous.get(px, py).expect("validated flow window")[0]; + let right = next.get(nx, ny).expect("validated flow window")[0]; + let difference = i32::from(left) - i32::from(right); + cost = cost.saturating_add((difference * difference) as u64); + } + } + cost +} + +/// Adaptive Gaussian background-model controls. +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct BackgroundModelOptions { + /// Exponential update factor in `(0, 1]`. + pub learning_rate: f32, + /// Foreground threshold measured in standard deviations. + pub sigma_threshold: f32, + /// Initial per-pixel variance. + pub initial_variance: f32, +} + +impl Default for BackgroundModelOptions { + fn default() -> Self { + Self { learning_rate: 0.02, sigma_threshold: 3.0, initial_variance: 225.0 } + } +} + +/// Stateful single-Gaussian grayscale background model. +#[derive(Clone, Debug, PartialEq)] +pub struct AdaptiveBackgroundModel { + width: usize, + height: usize, + mean: Vec, + variance: Vec, + options: BackgroundModelOptions, + frames_seen: u64, +} + +impl AdaptiveBackgroundModel { + /// Initializes the model from its first frame. + pub fn try_new( + frame: ImageView<'_, u8, 1>, + options: BackgroundModelOptions, + ) -> VisionResult { + validate_background_options(options)?; + let mean = packed_gray(frame).into_iter().map(f32::from).collect(); + Ok(Self { + width: frame.width(), + height: frame.height(), + mean, + variance: vec![options.initial_variance; frame.width() * frame.height()], + options, + frames_seen: 1, + }) + } + + /// Segments foreground and updates background statistics in one explicit step. + pub fn apply(&mut self, frame: ImageView<'_, u8, 1>) -> VisionResult { + if (frame.width(), frame.height()) != (self.width, self.height) { + return Err(VisionError::ShapeMismatch( + "background-model frame dimensions changed".to_owned(), + )); + } + let pixels = packed_gray(frame); + let mut mask = Vec::with_capacity(pixels.len()); + for (index, &pixel) in pixels.iter().enumerate() { + let value = f32::from(pixel); + let delta = value - self.mean[index]; + let threshold = self.options.sigma_threshold * self.variance[index].max(1.0).sqrt(); + let foreground = delta.abs() > threshold; + mask.push(u8::from(foreground)); + let alpha = if foreground { + self.options.learning_rate * 0.1 + } else { + self.options.learning_rate + }; + self.mean[index] += alpha * delta; + self.variance[index] = + ((1.0 - alpha) * self.variance[index] + alpha * delta * delta).max(1.0); + } + self.frames_seen = self.frames_seen.saturating_add(1); + let mask = BinaryMask::try_new(self.width, self.height, mask)?; + let foreground_ratio = mask.area() as f32 / (self.width * self.height).max(1) as f32; + Ok(BackgroundUpdate { mask, foreground_ratio, frame_index: self.frames_seen - 1 }) + } + + /// Returns the number of frames incorporated into the model. + #[must_use] + pub const fn frames_seen(&self) -> u64 { + self.frames_seen + } +} + +/// Output of one background-model update. +#[derive(Clone, Debug, PartialEq)] +pub struct BackgroundUpdate { + /// Binary foreground segmentation. + pub mask: BinaryMask, + /// Foreground area divided by image area. + pub foreground_ratio: f32, + /// Zero-based sequence index of the processed frame. + pub frame_index: u64, +} + +fn validate_background_options(options: BackgroundModelOptions) -> VisionResult<()> { + if !options.learning_rate.is_finite() + || !(0.0..=1.0).contains(&options.learning_rate) + || options.learning_rate == 0.0 + || !options.sigma_threshold.is_finite() + || options.sigma_threshold <= 0.0 + || !options.initial_variance.is_finite() + || options.initial_variance <= 0.0 + { + return Err(VisionError::InvalidParameter( + "invalid background-model learning/threshold/variance options".to_owned(), + )); + } + Ok(()) +} + +fn packed_gray(frame: ImageView<'_, u8, 1>) -> Vec { + let mut data = Vec::with_capacity(frame.width() * frame.height()); + for y in 0..frame.height() { + data.extend_from_slice(frame.row(y).expect("frame row in bounds")); + } + data +} + +/// Track lifecycle state. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum TrackState { + /// Track has not accumulated the configured hit count. + Tentative, + /// Track has accumulated enough consecutive associations. + Confirmed, +} + +/// One deterministic detection track. +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct ObjectTrack { + /// Monotonic track identifier. + pub id: u64, + /// Most recent bounding box. + pub bbox: BoundingBox2, + /// Model-defined class identifier. + pub class_id: i64, + /// Most recent detection score. + pub score: f32, + /// Frames since track creation. + pub age: u32, + /// Total successful associations. + pub hits: u32, + /// Consecutive frames without an association. + pub missed: u32, + /// Tentative/confirmed state. + pub state: TrackState, +} + +/// IoU tracker controls. +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct MultiObjectTrackerOptions { + /// Minimum IoU for same-class association. + pub iou_threshold: f32, + /// Consecutive misses retained before removal. + pub max_missed: u32, + /// Hits required to confirm a track. + pub min_confirmed_hits: u32, +} + +impl Default for MultiObjectTrackerOptions { + fn default() -> Self { + Self { iou_threshold: 0.3, max_missed: 3, min_confirmed_hits: 2 } + } +} + +/// Deterministic same-class greedy IoU multi-object tracker. +#[derive(Clone, Debug, PartialEq)] +pub struct MultiObjectTracker { + options: MultiObjectTrackerOptions, + next_id: u64, + tracks: Vec, +} + +impl MultiObjectTracker { + /// Creates an empty tracker. + pub fn try_new(options: MultiObjectTrackerOptions) -> VisionResult { + if !options.iou_threshold.is_finite() + || !(0.0..=1.0).contains(&options.iou_threshold) + || options.min_confirmed_hits == 0 + { + return Err(VisionError::InvalidParameter("invalid tracker options".to_owned())); + } + Ok(Self { options, next_id: 1, tracks: Vec::new() }) + } + + /// Associates one detection frame and returns current tracks ordered by id. + pub fn update(&mut self, detections: &[Detection]) -> VisionResult<&[ObjectTrack]> { + if detections + .iter() + .any(|detection| !detection.bbox.is_valid() || !detection.score.is_finite()) + { + return Err(VisionError::InvalidParameter( + "tracker detections must contain valid boxes and finite scores".to_owned(), + )); + } + for track in &mut self.tracks { + track.age = track.age.saturating_add(1); + track.missed = track.missed.saturating_add(1); + } + let mut used = vec![false; detections.len()]; + for track in &mut self.tracks { + let best = detections + .iter() + .enumerate() + .filter(|(index, detection)| !used[*index] && detection.class_id == track.class_id) + .map(|(index, detection)| (index, track.bbox.iou(detection.bbox))) + .filter(|(_, iou)| *iou >= self.options.iou_threshold) + .max_by(|left, right| { + left.1.total_cmp(&right.1).then_with(|| right.0.cmp(&left.0)) + }); + if let Some((index, _)) = best { + used[index] = true; + let detection = detections[index]; + track.bbox = detection.bbox; + track.score = detection.score; + track.hits = track.hits.saturating_add(1); + track.missed = 0; + if track.hits >= self.options.min_confirmed_hits { + track.state = TrackState::Confirmed; + } + } + } + for (index, &detection) in detections.iter().enumerate() { + if !used[index] { + let hits = 1; + self.tracks.push(ObjectTrack { + id: self.next_id, + bbox: detection.bbox, + class_id: detection.class_id, + score: detection.score, + age: 1, + hits, + missed: 0, + state: if hits >= self.options.min_confirmed_hits { + TrackState::Confirmed + } else { + TrackState::Tentative + }, + }); + self.next_id = self.next_id.saturating_add(1); + } + } + self.tracks.retain(|track| track.missed <= self.options.max_missed); + self.tracks.sort_by_key(|track| track.id); + Ok(&self.tracks) + } + + /// Borrows current tracks ordered by id. + #[must_use] + pub fn tracks(&self) -> &[ObjectTrack] { + &self.tracks + } +} + +/// Timestamped owned video frame. +#[cfg(feature = "video-adapters")] +#[derive(Clone, Debug, PartialEq)] +pub struct VideoFrame { + /// Monotonic source sequence. + pub sequence: u64, + /// Source timestamp in nanoseconds. + pub timestamp_ns: i64, + /// Owned typed image; adapters do not hide host/device copies. + pub image: Image, +} + +/// Pull-based video source implemented by optional codec/camera adapters. +#[cfg(feature = "video-adapters")] +pub trait VideoFrameSource { + /// Returns the next frame, or `None` at end of stream. + fn next_frame(&mut self) -> VisionResult>>; +} + +/// Deterministic in-memory source used for adapter conformance and replay. +#[cfg(feature = "video-adapters")] +#[derive(Clone, Debug, PartialEq)] +pub struct MemoryVideoSource { + frames: VecDeque>, +} + +#[cfg(feature = "video-adapters")] +impl MemoryVideoSource { + /// Creates a source and validates strictly increasing sequence/timestamps. + pub fn try_new(frames: Vec>) -> VisionResult { + if frames.windows(2).any(|pair| { + pair[0].sequence >= pair[1].sequence + || pair[0].timestamp_ns >= pair[1].timestamp_ns + || (pair[0].image.width(), pair[0].image.height()) + != (pair[1].image.width(), pair[1].image.height()) + }) { + return Err(VisionError::InvalidParameter( + "video frames need increasing sequence/time and stable dimensions".to_owned(), + )); + } + Ok(Self { frames: frames.into() }) + } +} + +#[cfg(feature = "video-adapters")] +impl VideoFrameSource for MemoryVideoSource { + fn next_frame(&mut self) -> VisionResult>> { + Ok(self.frames.pop_front()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn translated_texture( + width: usize, + height: usize, + dx: usize, + dy: usize, + ) -> (Image, Image) { + let mut previous = vec![0_u8; width * height]; + for y in 0..height { + for x in 0..width { + previous[y * width + x] = ((x * 17 + y * 29 + x * y * 3) % 251) as u8; + } + } + let mut next = vec![0_u8; width * height]; + for y in 0..height - dy { + for x in 0..width - dx { + next[(y + dy) * width + x + dx] = previous[y * width + x]; + } + } + ( + Image::try_new(width, height, previous).unwrap(), + Image::try_new(width, height, next).unwrap(), + ) + } + + #[test] + fn dense_flow_recovers_integer_translation() { + let (previous, next) = translated_texture(40, 32, 2, 1); + let flow = + dense_flow_block_match(previous.view(), next.view(), DenseFlowOptions::default()) + .unwrap(); + let center = flow.image().get(20, 16).unwrap(); + assert_eq!(center, &[2.0, 1.0]); + assert!(flow.image().get(0, 0).unwrap().iter().all(|value| value.is_nan())); + } + + #[test] + fn background_model_detects_new_foreground() { + let base = Image::::try_new(8, 8, vec![20; 64]).unwrap(); + let mut model = + AdaptiveBackgroundModel::try_new(base.view(), BackgroundModelOptions::default()) + .unwrap(); + let mut changed = vec![20_u8; 64]; + for y in 2..6 { + for x in 3..5 { + changed[y * 8 + x] = 200; + } + } + let frame = Image::::try_new(8, 8, changed).unwrap(); + let update = model.apply(frame.view()).unwrap(); + assert_eq!(update.mask.area(), 8); + assert_eq!(update.frame_index, 1); + } + + #[test] + fn tracker_preserves_id_and_expires_misses() { + let mut tracker = MultiObjectTracker::try_new(MultiObjectTrackerOptions { + max_missed: 1, + ..MultiObjectTrackerOptions::default() + }) + .unwrap(); + let detection = |x: f32| Detection { + bbox: BoundingBox2::try_new(x, 0.0, x + 10.0, 10.0).unwrap(), + score: 0.9, + class_id: 1, + }; + assert_eq!(tracker.update(&[detection(0.0)]).unwrap()[0].id, 1); + let tracks = tracker.update(&[detection(1.0)]).unwrap(); + assert_eq!(tracks[0].id, 1); + assert_eq!(tracks[0].state, TrackState::Confirmed); + assert_eq!(tracker.update(&[]).unwrap().len(), 1); + assert!(tracker.update(&[]).unwrap().is_empty()); + } + + #[cfg(feature = "video-adapters")] + #[test] + fn memory_source_preserves_sequence_and_end_of_stream() { + let frames = (0..3) + .map(|sequence| VideoFrame { + sequence, + timestamp_ns: sequence as i64 * 10, + image: Image::::try_new(2, 2, vec![sequence as u8; 4]).unwrap(), + }) + .collect(); + let mut source = MemoryVideoSource::try_new(frames).unwrap(); + for sequence in 0..3 { + assert_eq!(source.next_frame().unwrap().unwrap().sequence, sequence); + } + assert!(source.next_frame().unwrap().is_none()); + } +} diff --git a/crates/spatialrust/Cargo.toml b/crates/spatialrust/Cargo.toml index ab29ee8..6a73cbe 100644 --- a/crates/spatialrust/Cargo.toml +++ b/crates/spatialrust/Cargo.toml @@ -158,6 +158,8 @@ vision-detection = ["vision", "spatialrust-vision/detection"] vision-dense = ["vision-detection", "spatialrust-vision/dense"] vision-spatial = ["vision-dense", "camera-rgbd", "spatialrust-vision/spatial"] vision-ai-adapters = ["vision-preprocess", "vision-dense", "tensor-image", "spatialrust-vision/ai-adapters"] +vision-video = ["vision-dense", "spatialrust-vision/video"] +vision-video-adapters = ["vision-video", "spatialrust-vision/video-adapters"] vision-full = [ "vision-preprocess", "vision-warp", @@ -171,6 +173,7 @@ vision-full = [ "vision-dense", "vision-spatial", "vision-ai-adapters", + "vision-video-adapters", ] ai-vision-pipeline = ["ai", "vision-ai-adapters", "vision-spatial"] records = ["dep:spatialrust-records"] diff --git a/docs/API_STABILITY.md b/docs/API_STABILITY.md index edf82ac..954fbb7 100644 --- a/docs/API_STABILITY.md +++ b/docs/API_STABILITY.md @@ -63,7 +63,7 @@ until their individual 1.0 milestones. | Camera (`camera`, `camera-rgbd`) | Pinhole/Brown-Conrady models and explicit RGB-D conversion entry points are stable | | Image IO (`image-io-*`) | Bounded codecs, typed decoded pixels, and source metadata are provisional | | AI (`ai-*`) | Backend/session, named dynamic I/O, copy policy, I/O binding, mock backend, and ONNX Runtime adapter APIs are provisional | -| Vision (`vision-*`) | Base errors/borders, resize/filter entry points, detection/dense data contracts, and Feature2D data contracts are stable; geometry, stereo, optical flow, and AI adapters remain provisional | +| Vision (`vision-*`) | Base errors/borders, resize/filter entry points, detection/dense data contracts, and Feature2D data contracts are stable; geometry, stereo, optical flow, video, and AI adapters remain provisional | | Tensor (`tensor-*`) | Dtype/layout/device ownership, typed host storage, external host owner, and DLPack APIs are provisional | | Records (`records`) | Versioned `SpatialRecord`, schema compatibility/migration, and chunked record streams are provisional | | Arrow (`arrow-*`) | Arrow C Data/Stream/Device bridges for point clouds are provisional | diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 99f44d4..90ea565 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -80,6 +80,9 @@ PnP, sparse LK, stereo BM), warp, detection, dense-map, and spatial bridges in separate additive features. Geometry depends on `spatialrust-camera` only and does not pull Feature2D or dense-map types. CPU APIs never perform implicit device copies; future GPU/CUDA implementations belong behind explicit backend features. +Video algorithms depend on dense/detection contracts, while timestamped pull +sources are isolated behind `video-adapters`; native codec/camera runtimes stay +in future dedicated adapter crates/features. Its `imgproc-*` features share one border extrapolation contract; `filter2d` means correlation, while true convolution is an explicitly named operation. `spatialrust-tensor` is distinct from the point-cloud chunk iterator named diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 567a357..1d8caeb 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -382,7 +382,7 @@ uses the standard completion gates above and lands as one reviewable PR. | 103 | Complete | 101–102 | SIMD/parallel CPU kernel dispatch, reusable outputs, and measured allocation control | | 104 | Complete | 89, 101–103 | Texture-backed GPU Image v2 and device-resident resize/filter/edge/morphology chains | | 105 | Complete | 88, 101–102 | Mono/stereo/fisheye/hand-eye calibration and bundle-adjustment contracts | -| 106 | Planned | 92, 101–105 | Dense flow, tracking, background modeling, and feature-gated video stream adapters | +| 106 | Complete | 92, 101–105 | Dense flow, tracking, background modeling, and feature-gated video stream adapters | | 107 | Planned | 93, 101–106 | Stronger local features, robust tracking, and visual/RGB-D odometry integration | | 108 | Planned | 101–107 | Feature-gated computational photography and panorama stitching | | 109 | Planned | 97, 99, 101–108 | Bounded spatial execution graph with fusion, backpressure, and named transfer receipts | @@ -470,3 +470,18 @@ normal equations, and introduce no native optimizer dependency. Supplied rotations are checked for finite, right-handed orthonormal form. The first BA contract intentionally fixes calibrated camera poses and refines world points; joint pose/intrinsics optimization remains additive and provisional. + +### Epic 106 delivery slices + +| Slice | Status | Scope | Evidence | +| --- | --- | --- | --- | +| 106A | Complete | Dense deterministic integer block flow with invalid-border semantics | translated-texture test and OpenCV Farneback comparison | +| 106B | Complete | Adaptive single-Gaussian foreground modeling | new-object mask/ratio sequence test | +| 106C | Complete | Same-class IoU track lifecycle with monotonic IDs | confirmation, association, miss, and expiry test | +| 106D | Complete | Pull-based timestamped stream adapter contract | `VideoFrameSource` and deterministic `MemoryVideoSource` | +| 106E | Complete | Dedicated features, Python flow binding, and benchmark coverage | `vision-video[-adapters]`, QQVGA/QVGA Criterion | + +Core video algorithms stay runtime-free in `spatialrust-vision/video`. Codec, +camera, and network integrations implement `VideoFrameSource` behind additive +features; frames carry sequence/time explicitly and remain owned host images. +No adapter may hide a GPU transfer. diff --git a/notes/2026-07-15_epic106_video.md b/notes/2026-07-15_epic106_video.md new file mode 100644 index 0000000..1aa82f2 --- /dev/null +++ b/notes/2026-07-15_epic106_video.md @@ -0,0 +1,24 @@ +# Epic 106: video recognition substrate + +Epic 106 adds portable sequence algorithms to `spatialrust-vision` without +adding a codec or camera runtime to default builds. + +- Dense grayscale flow performs deterministic integer block matching and marks + pixels without a complete search window as NaN. +- `AdaptiveBackgroundModel` exposes learning rate, sigma threshold, initial + variance, foreground mask/ratio, and frame index. +- `MultiObjectTracker` uses deterministic same-class greedy IoU association, + monotonic IDs, tentative/confirmed states, and explicit miss expiry. +- `VideoFrameSource` pulls timestamped owned frames. `MemoryVideoSource` + validates monotonic sequence/time and stable dimensions for conformance tests. + +The OpenCV 4.10 receipt compares median motion on a deterministic translated +texture with Farneback flow. SpatialRust's integer median is gated within 0.01 +pixels of the known translation and within 0.5 pixels of OpenCV. External +FFmpeg/GStreamer/camera implementations remain additive adapters and cannot hide +host/device copies. + +The recorded `(3, 2)` translation produced a SpatialRust median of `(3.0, 2.0)` +with zero error. OpenCV Farneback produced +`(2.999943256378174, 1.99998140335083)`, for a cross-implementation maximum +error of `5.6743621826171875e-05` pixels.