diff --git a/Cargo.lock b/Cargo.lock index 6db0c65d..35bf2ce2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -399,6 +399,7 @@ dependencies = [ "ed25519-dalek 2.2.0", "flate2", "futures-core", + "futures-util", "hex", "http-body", "libc", diff --git a/crates/canopy-server/Cargo.toml b/crates/canopy-server/Cargo.toml index 93b6e72c..3dbe6650 100644 --- a/crates/canopy-server/Cargo.toml +++ b/crates/canopy-server/Cargo.toml @@ -23,6 +23,7 @@ cellule-store = { git = "https://github.com/crabbuild/cellule.git", rev = "16106 ed25519-dalek = "2" flate2 = "1.1" futures-core = "0.3" +futures-util = { version = "0.3", default-features = false, features = ["std"] } hex = "0.4" http-body = "1" object_store = "0.14.1" diff --git a/crates/canopy-server/src/admission.rs b/crates/canopy-server/src/admission.rs index 98099407..6afd514d 100644 --- a/crates/canopy-server/src/admission.rs +++ b/crates/canopy-server/src/admission.rs @@ -26,6 +26,10 @@ pub(crate) struct AccountAdmission { } impl AccountAdmission { + pub(crate) fn available(&self) -> usize { + self.total.available_permits() + } + pub(crate) fn new( limit: usize, total_capacity: &'static str, diff --git a/crates/canopy-server/src/deployment/mod.rs b/crates/canopy-server/src/deployment/mod.rs index 285962dd..ec5282a8 100644 --- a/crates/canopy-server/src/deployment/mod.rs +++ b/crates/canopy-server/src/deployment/mod.rs @@ -19,6 +19,14 @@ mod recovery; mod root; pub use recovery::WorkerConfig; +/// Incompatible repository deployment and local cache format. +pub const STORAGE_FORMAT: &str = "canopy-pack-v1"; + +/// Read-only admission before local reclamation, probes or Cell activation. +pub(crate) async fn validate_service_root(store: &Store, prefix: &Path) -> Result<()> { + root::validate_service(store, prefix).await +} + /// Application-wide admission shared by nodes and offline administration. #[derive(Clone)] pub struct Deployment { diff --git a/crates/canopy-server/src/deployment/recovery.rs b/crates/canopy-server/src/deployment/recovery.rs index 355e0e26..8300b9c9 100644 --- a/crates/canopy-server/src/deployment/recovery.rs +++ b/crates/canopy-server/src/deployment/recovery.rs @@ -168,7 +168,7 @@ impl Deployment { let (module, schema) = if entry.namespace() == directory::DIRECTORY { (DirectoryModule::NAME, directory::SCHEMA) } else if entry.namespace() == REPOSITORIES { - (RepositoryModule::NAME, include_str!("../schema.sql")) + (RepositoryModule::NAME, crate::REPOSITORY_SCHEMA) } else { return Err(Error::Release("unknown maintenance Cell namespace").into()); }; diff --git a/crates/canopy-server/src/deployment/root.rs b/crates/canopy-server/src/deployment/root.rs index a6497caf..28e01d9b 100644 --- a/crates/canopy-server/src/deployment/root.rs +++ b/crates/canopy-server/src/deployment/root.rs @@ -70,21 +70,74 @@ pub(super) struct RootClaim { token: ETag, } +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct RootEnvelope

{ + format: String, + purpose: P, +} + +fn encode(purpose: &RootPurpose) -> Result { + if matches!(purpose, RootPurpose::Backup { source, pin, .. } | RootPurpose::Restore { source, pin, .. } + if source.len() > 4096 || pin.len() > 4096) + { + return Err(Error::Backup("root reservation exceeds size limit")); + } + let body = Bytes::from(serde_json::to_vec(&RootEnvelope { + format: STORAGE_FORMAT.into(), + purpose, + })?); + if body.len() > 4096 { + return Err(Error::Backup("root reservation exceeds size limit")); + } + Ok(body) +} + fn path(root: &Path) -> Path { root.clone().join("canopy-root-v1.json") } pub(super) async fn load(store: &Store, root: &Path) -> Result> { match store.get_with_etag_bounded(&path(root), 4096).await { - Ok((bytes, token)) => Ok(Some(RootClaim { - purpose: serde_json::from_slice(&bytes)?, - token, - })), + Ok((bytes, token)) => { + let envelope: RootEnvelope = serde_json::from_slice(&bytes)?; + if envelope.format != STORAGE_FORMAT { + return Err(Error::Backup("unrecognized Canopy storage format")); + } + Ok(Some(RootClaim { + purpose: envelope.purpose, + token, + })) + } Err(StorageError::NotFound { .. }) => Ok(None), Err(error) => Err(error.into()), } } +pub(super) async fn validate_service(store: &Store, root: &Path) -> Result<()> { + let mut claim = load(store, root).await?; + if claim.is_none() + && ApplicationIdentityStore::new(store.clone(), root.clone()) + .load() + .await? + .is_some() + { + // A concurrent initializer reserves its marker before its identity. + claim = load(store, root).await?; + if claim.is_none() { + return Err(Error::Backup( + "destination already contains an application identity", + )); + } + } + if claim.is_some_and(|claim| !claim.purpose.permits_service()) { + return Err(Error::Backup( + "backup or unfinished restore prefix cannot serve", + )); + } + Ok(()) +} + pub(super) async fn reserve(store: &Store, root: &Path, purpose: RootPurpose) -> Result { if load(store, root).await?.is_none() && ApplicationIdentityStore::new(store.clone(), root.clone()) @@ -99,10 +152,7 @@ pub(super) async fn reserve(store: &Store, root: &Path, purpose: RootPurpose) -> "destination already contains an application identity", )); } - let body = Bytes::from(serde_json::to_vec(&purpose)?); - if body.len() > 4096 { - return Err(Error::Backup("root reservation exceeds size limit")); - } + let body = encode(&purpose)?; match store.create_strict_with_etag(&path(root), body).await { Ok(token) => Ok(RootClaim { purpose, token }), Err(error) => match load(store, root).await? { @@ -131,14 +181,7 @@ impl RootClaim { } RootPurpose::Service => return Err(Error::Backup("service root cannot finish a copy")), } - match store - .update( - &path(root), - Bytes::from(serde_json::to_vec(&next)?), - self.token, - ) - .await - { + match store.update(&path(root), encode(&next)?, self.token).await { Ok(_) => Ok(()), Err(error) => match load(store, root).await? { Some(current) if current.purpose == next => Ok(()), diff --git a/crates/canopy-server/src/deployment/tests.rs b/crates/canopy-server/src/deployment/tests.rs index 26ba872c..17b60658 100644 --- a/crates/canopy-server/src/deployment/tests.rs +++ b/crates/canopy-server/src/deployment/tests.rs @@ -10,6 +10,84 @@ type TestResult = std::result::Result<(), Box>; mod retained_maintenance; +#[tokio::test] +async fn root_reservation_limit_includes_the_format_envelope_before_any_write() -> TestResult { + let deployment = fixture()?; + let mut purpose = root::RootPurpose::Backup { + source: String::new(), + pin: uuid::Uuid::new_v4().to_string(), + complete: false, + }; + let previous_overhead = serde_json::to_vec(&purpose)?.len(); + if let root::RootPurpose::Backup { source, .. } = &mut purpose { + *source = "s".repeat(4096 - previous_overhead); + } + assert_eq!(serde_json::to_vec(&purpose)?.len(), 4096); + assert!(matches!( + root::reserve(deployment.layout.store(), &deployment.prefix, purpose).await, + Err(Error::Backup("root reservation exceeds size limit")) + )); + assert!( + root::load(deployment.layout.store(), &deployment.prefix) + .await? + .is_none() + ); + assert!(deployment.identities.load().await?.is_none()); + Ok(()) +} + +#[tokio::test] +async fn old_or_unknown_root_format_cannot_initialize_identity_or_release() -> TestResult { + for bytes in [ + br#"{"kind":"service"}"#.as_slice(), + br#"{"purpose":{"kind":"service"}}"#, + br#"{"format":"future-format","purpose":{"kind":"service"}}"#, + br#"{"format":"canopy-pack-v1","purpose":{"kind":"service"},"extra":true}"#, + ] { + let deployment = fixture()?; + let path = deployment.prefix.clone().join("canopy-root-v1.json"); + let original = bytes::Bytes::copy_from_slice(bytes); + deployment + .layout + .store() + .create_strict(&path, original.clone()) + .await?; + let before = deployment.layout.store().get_with_etag(&path).await?; + assert!( + root::load(deployment.layout.store(), &deployment.prefix) + .await + .is_err() + ); + assert!(deployment.initialize().await.is_err()); + assert!(deployment.identities.load().await?.is_none()); + assert!(deployment.releases.load().await?.is_none()); + assert_eq!( + deployment.layout.store().get_with_etag(&path).await?, + before + ); + } + Ok(()) +} + +#[tokio::test] +async fn packed_root_envelope_reuses_purpose_and_is_rejected_by_old_decoder() -> TestResult { + let deployment = fixture()?; + deployment.initialize().await?; + let path = deployment.prefix.clone().join("canopy-root-v1.json"); + let (bytes, _) = deployment.layout.store().get_with_etag(&path).await?; + assert_eq!( + serde_json::from_slice::(&bytes)?, + serde_json::json!({ + "format": "canopy-pack-v1", "purpose": { "kind": "service" } + }) + ); + // The old decoder is exactly the existing tagged RootPurpose type. + assert!(serde_json::from_slice::(&bytes).is_err()); + deployment.initialize().await?; + deployment.require_ready().await?; + Ok(()) +} + fn fixture() -> std::result::Result> { let app = CanopyApplication::compile(build_descriptor( include_bytes!("../../../../Cargo.lock"), diff --git a/crates/canopy-server/src/git_cache/artifacts.rs b/crates/canopy-server/src/git_cache/artifacts.rs index e461dfe2..c077aa4a 100644 --- a/crates/canopy-server/src/git_cache/artifacts.rs +++ b/crates/canopy-server/src/git_cache/artifacts.rs @@ -6,15 +6,27 @@ use canopy_object_storage::artifact::{ArtifactKind, ArtifactStore}; struct Writer { file: File, _cache: Arc, + _owner: crate::git_objects::ReadOwner, } impl GitCache { /// The isolated verifier calls this exactly once on its fresh private cache. /// Reserve the complete pair before creating files or reading the provider. /// No second pack copy or blob-as-artifact wrapper is involved. + #[cfg(test)] pub(crate) async fn download_native( self: &Arc, store: &ArtifactStore, descriptor: NativePackDescriptor, + ) -> Result<(), MetadataError> { + self.download_native_owned(store, descriptor, Arc::new(())) + .await + } + + pub(crate) async fn download_native_owned( + self: &Arc, + store: &ArtifactStore, + descriptor: NativePackDescriptor, + owner: crate::git_objects::ReadOwner, ) -> Result<(), MetadataError> { descriptor .validate(store.repository(), self.object_format) @@ -23,7 +35,9 @@ impl GitCache { return Err(MetadataError::Integrity); } let cache = Arc::clone(self); + let admission = Arc::clone(&owner); tokio::task::spawn_blocking(move || { + let _owner = admission; let size = descriptor .pack .size @@ -38,6 +52,7 @@ impl GitCache { (ArtifactKind::Index, descriptor.index, "idx"), ] { let cache = Arc::clone(self); + let admission = Arc::clone(&owner); let mut writer = tokio::task::spawn_blocking(move || { let path = cache.git_dir().join(format!( "objects/pack/pack-{}.{}", @@ -47,6 +62,7 @@ impl GitCache { Ok::<_, MetadataError>(Writer { file: File::create_new(path)?, _cache: cache, + _owner: admission, }) }) .await??; diff --git a/crates/canopy-server/src/git_cache/cleanup.rs b/crates/canopy-server/src/git_cache/cleanup.rs index 62e1fa66..0c4f26fc 100644 --- a/crates/canopy-server/src/git_cache/cleanup.rs +++ b/crates/canopy-server/src/git_cache/cleanup.rs @@ -8,6 +8,7 @@ pub(super) struct Cleanup { pub(super) path: PathBuf, pub(super) reservation: Option, pub(super) objects: Option>, + pub(super) owner: Option, } impl Cleanup { @@ -45,6 +46,7 @@ impl Cleanup { // Successful removal permits normal field teardown. self.reservation.take(); self.objects.take(); + self.owner.take(); }).await; return; } @@ -68,6 +70,9 @@ impl Drop for Cleanup { if let Some(reservation) = self.reservation.take() { std::mem::forget(reservation); } + if let Some(owner) = self.owner.take() { + std::mem::forget(owner); + } if let Some(objects) = self.objects.take() { std::mem::forget(objects); } diff --git a/crates/canopy-server/src/git_cache/maintenance.rs b/crates/canopy-server/src/git_cache/maintenance.rs index 09028b84..4193a43a 100644 --- a/crates/canopy-server/src/git_cache/maintenance.rs +++ b/crates/canopy-server/src/git_cache/maintenance.rs @@ -1,9 +1,13 @@ //! Immutable cache generations: never repack/delete files beneath an active reader. use super::*; +#[cfg(test)] use crate::git_http::{GitHttpError, GitProcess, WORKER_DEADLINE, read_bounded}; +#[cfg(test)] use std::process::Stdio; +#[cfg(test)] use tokio::io::AsyncWriteExt; +#[cfg(test)] fn worker_error(error: GitHttpError) -> CacheError { io::Error::other(error).into() } @@ -247,6 +251,7 @@ impl GitCache { } /// Reuse only packs whose *every* object was verified and durably recorded /// during this ingestion. Extra/unverified objects disable this optimization. + #[cfg(test)] pub(crate) async fn retain_verified_packs( self: &Arc, source: Arc, @@ -318,6 +323,7 @@ impl GitCache { /// Enumerate a captured cache into a new self-contained pack. Old objects /// and packs are untouched; dropping the last old reader reclaims them. + #[cfg(test)] pub(crate) async fn repacked( self: &Arc, root: PathBuf, @@ -346,7 +352,6 @@ impl GitCache { next.reservation()? .try_grow(reserve) .map_err(io::Error::other)?; - let coverage = self.prepared.lock().await.clone(); let durable = self .durable_packs .read() @@ -497,7 +502,6 @@ impl GitCache { result.map_err(worker_error)?; next.pack_files .store(1, std::sync::atomic::Ordering::Relaxed); - *next.prepared.lock().await = coverage; *next .durable_packs .write() @@ -506,6 +510,7 @@ impl GitCache { } } +#[cfg(test)] async fn finish( process: &mut GitProcess, stderr: Vec, diff --git a/crates/canopy-server/src/git_cache/mod.rs b/crates/canopy-server/src/git_cache/mod.rs index e723b76f..77009cf5 100644 --- a/crates/canopy-server/src/git_cache/mod.rs +++ b/crates/canopy-server/src/git_cache/mod.rs @@ -1,7 +1,7 @@ //! Disposable Git files charged to the node's shared disk budget. use std::{ - collections::{BTreeMap, BTreeSet, HashSet}, + collections::HashSet, fs::{self, File, OpenOptions}, io::{self, BufWriter, Write}, path::{Path, PathBuf}, @@ -9,20 +9,26 @@ use std::{ }; use cellule_ltx::{DiskBudget, DiskReservation, LtxError}; +#[cfg(test)] use flate2::{Compression, write::ZlibEncoder}; +#[cfg(test)] +use std::collections::BTreeMap; use tokio::sync::Mutex; use crate::{ - ObjectKind, RefExpectation, blob::{LargeBlobError, LargeBlobRead}, - object_id, refs::valid_ref_name, }; +#[cfg(test)] +use crate::object_id; +use crate::{ObjectKind, RefExpectation}; + pub(crate) const CACHE_PREFIX: &str = "canopy-git-"; mod artifacts; mod cleanup; +mod serving_refs; #[derive(Debug, thiserror::Error)] pub enum CacheError { @@ -50,22 +56,30 @@ pub(crate) enum ReceiveHook { PreReceive, } +/// Creation work and retained file admission have different lifetimes. Only the +/// cleanup charge belongs in an idle cached file; work may pin a generation. +pub(crate) struct CacheOwnership { + pub(crate) work: crate::git_objects::ReadOwner, + pub(crate) cleanup: Option, +} + pub(crate) struct GitCache { pub(crate) object_format: crate::ObjectFormat, pub(crate) native: crate::native_resources::NativeScope, directory: tempfile::TempDir, reservation: Option, + cleanup_owner: Option, objects: Option>, // Only durable hydration writes this cache. Stripe by OID so concurrent // fetches share a completed loose object without serializing all objects. + #[cfg(test)] object_writes: OnceLock<[Arc>; 64]>, packed: RwLock>, durable_packs: RwLock>, pub(crate) selection: Mutex<()>, - pub(crate) prepared: Mutex>, + #[cfg(test)] pub(crate) loose_objects: std::sync::atomic::AtomicU64, pub(crate) pack_files: std::sync::atomic::AtomicU64, - pub(crate) hydrating: std::sync::atomic::AtomicU64, pub(crate) write_generation: std::sync::atomic::AtomicU64, } @@ -87,12 +101,37 @@ impl GitCache { object_format: crate::ObjectFormat, objects: Option>, native: crate::native_resources::NativeScope, + ) -> Result, CacheError> { + Self::create_owned( + root, + budget, + head, + object_format, + objects, + native, + CacheOwnership { + work: Arc::new(()), + cleanup: None, + }, + ) + .await + } + + pub(crate) async fn create_owned( + root: PathBuf, + budget: DiskBudget, + head: &str, + object_format: crate::ObjectFormat, + objects: Option>, + native: crate::native_resources::NativeScope, + owner: CacheOwnership, ) -> Result, CacheError> { if !crate::default_branch::valid_default_branch(head) { return Err(CacheError::InvalidHead); } let head = format!("ref: {head}\n"); tokio::task::spawn_blocking(move || { + let CacheOwnership { work: _owner, cleanup } = owner; let cache = Arc::new(Self { object_format, native, @@ -100,15 +139,16 @@ impl GitCache { // absolute even when the node's data directory is relative. directory: tempfile::Builder::new().prefix(CACHE_PREFIX).tempdir_in(fs::canonicalize(root)?)?, reservation: Some(budget.try_reserve(0)?), + cleanup_owner: cleanup, objects, - object_writes: OnceLock::new(), + #[cfg(test)] + object_writes: OnceLock::new(), packed: RwLock::new(Vec::new()), durable_packs: RwLock::new(HashSet::new()), selection: Mutex::new(()), - prepared: Mutex::new(BTreeSet::new()), + #[cfg(test)] loose_objects: std::sync::atomic::AtomicU64::new(0), pack_files: std::sync::atomic::AtomicU64::new(0), - hydrating: std::sync::atomic::AtomicU64::new(0), write_generation: std::sync::atomic::AtomicU64::new(0), }); for directory in ["objects/info", "objects/pack", "refs/heads", "refs/tags", "hooks"] { @@ -142,24 +182,13 @@ impl GitCache { self.root().join("repo.git") } - pub(crate) fn object_cache(self: &Arc) -> Arc { - self.objects - .as_ref() - .map_or_else(|| Arc::clone(self), Arc::clone) - } - - pub(crate) fn hydration_guard(self: &Arc) -> HydrationGuard { - self.hydrating - .fetch_add(1, std::sync::atomic::Ordering::SeqCst); - HydrationGuard(Arc::clone(self)) - } - fn reservation(&self) -> io::Result<&DiskReservation> { self.reservation .as_ref() .ok_or_else(|| io::Error::other("Git cache accounting is closed")) } + #[cfg(test)] pub(crate) fn bytes(&self) -> io::Result { Ok(self.reservation()?.bytes()) } @@ -179,6 +208,7 @@ impl GitCache { self.writer(Path::new(relative))?.write_all(bytes) } + #[cfg(test)] pub(crate) async fn missing_objects( self: &Arc, ids: Vec, @@ -196,6 +226,7 @@ impl GitCache { .await? } + #[cfg(test)] fn object_path(&self, oid: crate::ObjectId) -> PathBuf { let hex = hex::encode(oid); self.git_dir() @@ -204,6 +235,7 @@ impl GitCache { .join(&hex[2..]) } + #[cfg(test)] fn object_present(&self, oid: crate::ObjectId) -> io::Result { for index in self .packed @@ -226,6 +258,7 @@ impl GitCache { } } + #[cfg(test)] fn object_write_lock(&self, oid: crate::ObjectId) -> Arc> { let stripes = self .object_writes @@ -233,6 +266,7 @@ impl GitCache { Arc::clone(&stripes[oid[0] as usize % stripes.len()]) } + #[cfg(test)] fn object_writer( self: &Arc, oid: crate::ObjectId, @@ -291,6 +325,7 @@ impl GitCache { .await? } + #[cfg(test)] pub(crate) async fn store_object( self: &Arc, oid: crate::ObjectId, @@ -325,18 +360,21 @@ impl GitCache { .await? } + #[cfg(test)] pub(crate) async fn store_blob( self: &Arc, reader: LargeBlobRead, ) -> Result<(), CacheError> { self.store_blob_reader(BlobReader::External(reader)).await } + #[cfg(test)] pub(crate) async fn store_native_blob( self: &Arc, reader: crate::pack_store::NativePackedRead, ) -> Result<(), CacheError> { self.store_blob_reader(BlobReader::Packed(reader)).await } + #[cfg(test)] async fn store_blob_reader(self: &Arc, mut reader: BlobReader) -> Result<(), CacheError> { let (oid, size) = reader.metadata(); let write = self.object_write_lock(oid).lock_owned().await; @@ -387,6 +425,7 @@ impl GitCache { .await? } + #[cfg(test)] pub(crate) async fn store_refs( self: &Arc, refs: &BTreeMap, @@ -420,8 +459,16 @@ impl GitCache { /// Measures native Git's completed writes before the gateway can publish refs. pub(crate) async fn reconcile(self: &Arc) -> Result<(), CacheError> { + self.reconcile_owned(Arc::new(())).await + } + + pub(crate) async fn reconcile_owned( + self: &Arc, + owner: crate::git_objects::ReadOwner, + ) -> Result<(), CacheError> { let cache = Arc::clone(self); tokio::task::spawn_blocking(move || { + let _owner = owner; cache.reservation()?.resize(tree_bytes(cache.root())?)?; Ok(()) }) @@ -430,6 +477,7 @@ impl GitCache { /// Physical index entries, including duplicates across packs. This is an /// admission/telemetry bound, never a proof of canonical object coverage. + #[cfg(test)] pub(crate) fn indexed_entries(&self) -> u64 { self.packed .read() @@ -444,6 +492,7 @@ impl GitCache { ) } + #[cfg(test)] fn register_index(&self, path: &Path) -> io::Result<()> { self.register_checked_index(crate::git_format::pack_index::PackIndex::open( path, @@ -487,6 +536,7 @@ impl Drop for GitCache { path: self.root().to_path_buf(), reservation: self.reservation.take(), objects: self.objects.take(), + owner: self.cleanup_owner.take(), } .defer(); return; @@ -502,6 +552,9 @@ impl Drop for GitCache { if let Some(objects) = self.objects.take() { std::mem::forget(objects); } + if let Some(owner) = self.cleanup_owner.take() { + std::mem::forget(owner); + } // TempDir must not retry deletion after a worker fence rejected it. // Startup reclaims this directory once all descendants have exited. self.directory.disable_cleanup(true); @@ -558,19 +611,12 @@ mod tests; mod maintenance; -pub(crate) struct HydrationGuard(Arc); -impl Drop for HydrationGuard { - fn drop(&mut self) { - self.0 - .hydrating - .fetch_sub(1, std::sync::atomic::Ordering::SeqCst); - } -} - +#[cfg(test)] enum BlobReader { External(LargeBlobRead), Packed(crate::pack_store::NativePackedRead), } +#[cfg(test)] impl BlobReader { fn metadata(&self) -> (crate::ObjectId, u64) { match self { diff --git a/crates/canopy-server/src/git_cache/serving_refs.rs b/crates/canopy-server/src/git_cache/serving_refs.rs new file mode 100644 index 00000000..7d7feffb --- /dev/null +++ b/crates/canopy-server/src/git_cache/serving_refs.rs @@ -0,0 +1,63 @@ +//! Count-bounded ref pages streamed into one unpublished, admitted native cache. +use super::*; +use crate::git_objects::ReadOwner; + +pub(crate) struct ServingRefsWriter { + output: BufWriter, + last: String, + _owner: ReadOwner, +} +impl GitCache { + pub(crate) async fn serving_refs( + self: &Arc, + owner: ReadOwner, + ) -> Result { + let cache = self.clone(); + tokio::task::spawn_blocking(move || { + let mut output = BufWriter::new(cache.writer(Path::new("packed-refs"))?); + output.write_all(b"# pack-refs with: sorted\n")?; + Ok(ServingRefsWriter { + output, + last: String::new(), + _owner: owner, + }) + }) + .await? + } +} +impl ServingRefsWriter { + pub(crate) async fn append( + mut self, + page: Vec<(String, RefExpectation)>, + ) -> Result { + if page.len() > crate::refs::REF_PAGE_SIZE { + return Err(CacheError::InvalidHead); + } + tokio::task::spawn_blocking(move || { + for (name, state) in page { + let Some(oid) = state.oid else { + return Err(CacheError::InvalidHead); + }; + if name <= self.last + || !valid_ref_name(&name) + || oid.is_zero() + || oid.format() != self.output.get_ref().cache.object_format + { + return Err(CacheError::InvalidHead); + } + writeln!(self.output, "{} {name}", hex::encode(oid))?; + self.last = name; + } + Ok(self) + }) + .await? + } + pub(crate) async fn finish(mut self) -> Result<(), CacheError> { + tokio::task::spawn_blocking(move || { + self.output.flush()?; + self.output.get_ref().file.sync_all()?; + Ok(()) + }) + .await? + } +} diff --git a/crates/canopy-server/src/git_cache/tests.rs b/crates/canopy-server/src/git_cache/tests.rs index b25a8df2..8fba281e 100644 --- a/crates/canopy-server/src/git_cache/tests.rs +++ b/crates/canopy-server/src/git_cache/tests.rs @@ -629,7 +629,6 @@ async fn incomplete_pack_extracts_verified_large_blobs_without_admitting_foreign pack: reader.upload(pack).await?, index: reader.upload(index).await?, approved: false, - covered_through: 0, }; let target = GitCache::create( root.path().into(), diff --git a/crates/canopy-server/src/git_gateway/branch_policy.rs b/crates/canopy-server/src/git_gateway/branch_policy.rs index 725dc51b..c6c12352 100644 --- a/crates/canopy-server/src/git_gateway/branch_policy.rs +++ b/crates/canopy-server/src/git_gateway/branch_policy.rs @@ -23,6 +23,17 @@ pub(super) enum PushCommands { } impl PushCommands { + pub(super) fn names(&self) -> Vec { + match self { + Self::Parsed { updates, .. } => { + let mut names: Vec<_> = updates.iter().map(|update| update.name.clone()).collect(); + names.sort(); + names + } + Self::OtherMedia | Self::Limited => Vec::new(), + } + } + pub(super) async fn read( request: &GitHttpRequest, format: crate::ObjectFormat, diff --git a/crates/canopy-server/src/git_gateway/candidates/mod.rs b/crates/canopy-server/src/git_gateway/candidates/mod.rs index 292e9b33..91b65ade 100644 --- a/crates/canopy-server/src/git_gateway/candidates/mod.rs +++ b/crates/canopy-server/src/git_gateway/candidates/mod.rs @@ -52,7 +52,7 @@ impl GitGateway { }) { return Ok(CandidateOutcome::Conflict); } - let cached = self.build_cache(self.cell_refs().await?, true).await?; + let cached = self.build_cache(actor, &[]).await?; let result = prepare_native(&self.repository, &cached.backend, &candidate).await?; if !valid_result(&result) { return Err(GitHttpError::TooLarge.into()); @@ -67,7 +67,7 @@ impl GitGateway { new_oid: Some(parse_oid(oid)?), }], }; - self.persist_objects(&cached.backend, &cached.snapshot.refs, &plan) + self.persist_objects(&cached.backend, &cached.refs, &plan) .await?; self.repository .prepare_graph(&plan) @@ -220,13 +220,13 @@ fn merge_output(bytes: &[u8]) -> Result<(String, Vec<&[u8]>), GatewayError> { Ok((tree.into(), paths)) } -struct Output { - status: ExitStatus, - stdout: Vec, +pub(super) struct Output { + pub(super) status: ExitStatus, + pub(super) stdout: Vec, stderr: Vec, } impl Output { - fn error(self) -> GatewayError { + pub(super) fn error(self) -> GatewayError { GitHttpError::GitExit { status: self.status, stderr: String::from_utf8_lossy(&self.stderr).into_owned(), @@ -234,7 +234,7 @@ impl Output { .into() } } -async fn run( +pub(super) async fn run( backend: &GitHttpBackend, args: &[&str], input: &[u8], diff --git a/crates/canopy-server/src/git_gateway/discovery.rs b/crates/canopy-server/src/git_gateway/discovery.rs index 4d65fac8..3b38b46f 100644 --- a/crates/canopy-server/src/git_gateway/discovery.rs +++ b/crates/canopy-server/src/git_gateway/discovery.rs @@ -1,5 +1,4 @@ use super::*; -use std::time::Instant; pub(super) async fn is_ref_discovery(request: &GitHttpRequest) -> Result { if request.method == "GET" && request.path_info == "/repo.git/info/refs" { @@ -50,67 +49,6 @@ fn ls_refs(mut bytes: &[u8]) -> bool { } } -impl GitGateway { - pub(super) async fn discovery_cache( - &self, - snapshot: RefSnapshot, - ) -> Result { - let started = Instant::now(); - let backend = GitHttpBackend::initialize( - self.scratch_root.clone(), - self.disk_budget.clone(), - &snapshot.head, - self.repository.object_format(), - self.native.clone(), - ) - .await? - .with_nonce(self.certificate_nonce().await?); - let mut pending: BTreeSet<_> = snapshot - .refs - .values() - .filter_map(|state| state.oid) - .collect(); - let mut visited = BTreeSet::new(); - let mut stats = Hydration::default(); - while !pending.is_empty() { - let ids: Vec<_> = pending.iter().take(MAX_OBJECTS).copied().collect(); - let page = self - .repository - .selected_objects(&ids) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - if page.is_empty() { - return Err(GatewayError::MalformedCache); - } - for object in page { - pending.remove(&object.oid); - visited.insert(object.oid); - let tag = object.kind == ObjectKind::Tag; - let target = self - .cache_object(&backend.cache, object, &mut stats) - .await?; - if tag { - let target = target.ok_or(GatewayError::MalformedCache)?; - if !visited.contains(&target) { - pending.insert(target); - } - } - } - } - // Native discovery checks ref target existence and peels tag chains. - // Commit parents and tree contents are needed only by later transfer RPCs. - backend.cache.store_refs(&snapshot.refs).await?; - tracing::debug!( - repository = %hex::encode(self.repository.repository_id()), - objects = stats.objects, - bytes = stats.bytes, - elapsed_seconds = started.elapsed().as_secs_f64(), - "prepared Git ref discovery" - ); - Ok(backend) - } -} - #[cfg(test)] mod tests { use super::ls_refs; diff --git a/crates/canopy-server/src/git_gateway/fetch.rs b/crates/canopy-server/src/git_gateway/fetch.rs index 5b5d4a69..85318d67 100644 --- a/crates/canopy-server/src/git_gateway/fetch.rs +++ b/crates/canopy-server/src/git_gateway/fetch.rs @@ -1,9 +1,4 @@ use super::*; -use cellule_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; - -// Correlate each ancestor lookup so SQLite can stop at the first live ref, -// without materializing the entire reverse graph or scanning all refs. -const REACHABLE_WANT: &str = "WITH RECURSIVE ancestors(oid) AS (VALUES (?1) UNION SELECT e.parent FROM object_edges e JOIN ancestors a ON e.child = a.oid) SELECT g.generation, EXISTS (SELECT 1 FROM ancestors a WHERE EXISTS (SELECT 1 FROM refs r WHERE r.oid = a.oid)) FROM ref_generation g WHERE g.singleton = 1"; pub(super) struct FetchRequest { pub(super) wants: BTreeSet, @@ -104,339 +99,43 @@ fn check_filter_policy(value: &str) -> Result<(), InputError> { } impl GitGateway { - pub(super) async fn prepare_fetch( - &self, - cached: &CachedRepository, - request: FetchRequest, + /// Validate every wanted object against the chosen live refs' complete + /// certified closure. Native pack presence never authorizes a guessed OID. + pub(crate) async fn validate_wants( + workspace: &crate::packs::publication::NativeWorkspace, + wants: &BTreeSet, ) -> Result<(), GatewayError> { - if request.wants.is_empty() { - return Ok(()); - } - let started = std::time::Instant::now(); - // The shared object cache publishes loose objects atomically and - // coordinates duplicate OID writes. Hold the gateway lock only long - // enough to borrow it; slow fetches must not queue behind each other. - let shared = cached.backend.cache.object_cache(); - let _hydrating = shared.hydration_guard(); - let through = { - let objects = self.objects.lock().await; - objects - .as_ref() - .filter(|objects| Arc::ptr_eq(&objects.cache, &shared)) - .map_or(0, |objects| objects.through) - }; - if self - .repository - .object_high_water() - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))? - .output - <= through + if wants + .iter() + .any(|id| id.is_zero() || id.format() != workspace.object_format()) { - shared - .prepared - .lock() - .await - .extend(request.wants.iter().map(|oid| (*oid, true))); - return Ok(()); + return Err(GatewayError::UnreachableWant); } - let roots: Vec<_> = request.wants.iter().copied().collect(); - let unfiltered = request.filter.is_none(); - if roots.iter().all(|oid| { - shared.prepared.try_lock().ok().is_some_and(|prepared| { - prepared.contains(&(*oid, true)) - || (request.filter.as_deref() == Some("blob:none") - && prepared.contains(&(*oid, false))) - }) - }) { - return Ok(()); - } - let _selection = shared.selection.lock().await; - // A concurrent cold request may have completed while we waited. - if (unfiltered || request.filter.as_deref() == Some("blob:none")) - && roots.iter().all(|oid| { - shared.prepared.try_lock().ok().is_some_and(|prepared| { - prepared.contains(&(*oid, true)) - || (!unfiltered && prepared.contains(&(*oid, false))) - }) - }) - { - return Ok(()); - } - self.hydrate_selected(&shared, request.wants).await?; - // The certified Cell graph already names every reachable blob. A full - // fetch can hydrate those bodies during the structural walk and avoid - // a second native traversal over the same cold history. - self.hydrate_structure(&shared, &roots, unfiltered, through) - .await?; - if unfiltered || request.filter.as_deref() == Some("blob:none") { - shared - .prepared - .lock() + // Even an empty discovery/negotiation group rechecks current access. + if wants.is_empty() { + workspace + .contains(&[]) .await - .extend(roots.iter().map(|oid| (*oid, unfiltered))); - // Per-root preparation above is a coverage certificate. Physical - // index counts include cross-pack duplicates and cannot certify a - // whole-repository watermark. Durable covering packs are handled by - // hydrate_selected using their committed covered_through metadata. + .map_err(|e| GatewayError::Cell(Box::new(e)))?; return Ok(()); } - // Use the same native filter as upload-pack. Structure is present, so - // tree/type filters can omit missing blobs without reading their bodies. - // Git conservatively includes missing blobs under size filters. - let mut walk = crate::git_objects::GitObjectWalk::missing( - &cached.backend.git_dir(), - roots, - request.filter.as_deref(), - &cached.backend.cache.native, - )?; - let mut stats = Hydration::default(); + let mut ids = wants.iter().copied(); loop { - let mut ids = Vec::with_capacity(MAX_OBJECTS); - for _ in 0..MAX_OBJECTS { - let Some(oid) = walk.next().await? else { - break; - }; - ids.push(oid); - } - if ids.is_empty() { + let page: Vec<_> = ids + .by_ref() + .take(crate::packs::metadata::PAGE_OBJECTS) + .collect(); + if page.is_empty() { break; } - self.hydrate_objects(&shared, ids, &mut stats).await?; - } - walk.finish().await?; - tracing::debug!( - repository = %hex::encode(self.repository.repository_id()), - objects = stats.objects, - bytes = stats.bytes, - elapsed_seconds = started.elapsed().as_secs_f64(), - "prepared reachable Git blobs" - ); - Ok(()) - } - - pub(super) async fn fetch_cache( - &self, - wants: &BTreeSet, - ) -> Result, GatewayError> { - if let Some(cached) = self.current_cache().await? { - self.validate_wants(&cached.snapshot, wants).await?; - return Ok(cached); - } - let live_refs = self.cell_refs().await?; - self.validate_wants(&live_refs, wants).await?; - let mut cache = self.cache.lock().await; - if cache - .as_ref() - .is_none_or(|cached| cached.snapshot != live_refs) - { - *cache = None; - *cache = Some(Arc::new(self.build_cache(live_refs, false).await?)); - } - Ok(Arc::clone( - cache.as_ref().ok_or(GatewayError::MalformedCache)?, - )) - } - - pub(super) async fn validate_wants( - &self, - snapshot: &RefSnapshot, - wants: &BTreeSet, - ) -> Result<(), GatewayError> { - // Native reachable-want validation walks commits only. Reverse edges - // certified by the Cell also fence trees and blobs, including cached - // objects retained after a ref deletion. Generation binds every batch. - let ids: Vec<_> = wants.iter().collect(); - for ids in ids.chunks(MAX_OBJECTS) { - let result = self - .repository - .sql - .query( - None, - SqlBatch { - statements: ids - .iter() - .map(|oid| SqlStatement { - sql: REACHABLE_WANT.into(), - parameters: vec![SqlValue::Blob(oid.to_vec())], - }) - .collect(), - }, - ) + if workspace + .contains(&page) .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - for set in result.output { - let Some([SqlValue::Integer(generation), SqlValue::Integer(reachable)]) = - set.rows.first().map(Vec::as_slice) - else { - return Err(GatewayError::MalformedCache); - }; - if *generation != snapshot.generation { - return Err(GatewayError::RefSnapshotBusy); - } - if *reachable != 1 { - return Err(GatewayError::UnreachableWant); - } - } - } - Ok(()) - } - - pub(super) async fn hydrate_selected( - &self, - cache: &Arc, - mut pending: BTreeSet, - ) -> Result<(), GatewayError> { - let mut visited = BTreeSet::new(); - let mut stats = Hydration::default(); - while !pending.is_empty() { - let ids: Vec<_> = pending.iter().take(MAX_OBJECTS).copied().collect(); - // Advertisements peel tags even with blob filtering. Follow only - // tag edges here; ordinary tree descendants stay omitted. - let placeholders = vec!["?"; ids.len()].join(","); - let result = self.repository.sql.query(None, SqlBatch { statements: vec![SqlStatement { - sql: format!("SELECT e.child FROM object_edges e JOIN objects o ON o.oid = e.parent WHERE o.kind = 'tag' AND e.parent IN ({placeholders})"), - parameters: ids.iter().map(|oid| SqlValue::Blob(oid.to_vec())).collect(), - }] }).await.map_err(|error| GatewayError::Cell(Box::new(error)))?; - for oid in &ids { - pending.remove(oid); - visited.insert(*oid); - } - for row in &result - .output - .first() - .ok_or(GatewayError::MalformedCache)? - .rows + .map_err(|e| GatewayError::Cell(Box::new(e)))? + .iter() + .any(|present| !present) { - let [SqlValue::Blob(oid)] = row.as_slice() else { - return Err(GatewayError::MalformedCache); - }; - let oid = oid - .as_slice() - .try_into() - .map_err(|_| GatewayError::MalformedCache)?; - if !visited.contains(&oid) { - pending.insert(oid); - } - } - self.hydrate_objects(cache, ids, &mut stats).await?; - } - tracing::debug!( - objects = stats.objects, - bytes = stats.bytes, - "hydrated explicit Git objects" - ); - Ok(()) - } - - async fn hydrate_structure( - &self, - cache: &Arc, - roots: &[crate::ObjectId], - include_blobs: bool, - through: i64, - ) -> Result<(), GatewayError> { - let mut pending: BTreeSet<_> = roots.iter().copied().collect(); - let mut visited = BTreeSet::new(); - let mut stats = Hydration::default(); - while !pending.is_empty() { - let ids: Vec<_> = pending.iter().take(MAX_OBJECTS).copied().collect(); - for oid in &ids { - pending.remove(oid); - visited.insert(*oid); - } - let placeholders = vec!["?"; ids.len()].join(","); - let kind_filter = if include_blobs { - "" - } else { - "AND o.kind != 'blob'" - }; - let mut after_parent = Vec::new(); - let mut after_child = Vec::new(); - loop { - // Certified edges supply only reachable structure. Page both - // parents and children so a large tree stays within SQL wire bounds. - let mut parameters: Vec<_> = - ids.iter().map(|oid| SqlValue::Blob(oid.to_vec())).collect(); - parameters.extend([ - SqlValue::Integer(through), - SqlValue::Blob(after_parent.clone()), - SqlValue::Blob(after_parent.clone()), - SqlValue::Blob(after_child.clone()), - SqlValue::Integer(MAX_OBJECTS as i64), - ]); - let result = self.repository.sql.query(None, SqlBatch { statements: vec![SqlStatement { - sql: format!("SELECT e.parent, e.child, o.kind FROM object_edges e JOIN objects o ON o.oid = e.child JOIN objects p ON p.oid = e.parent WHERE e.parent IN ({placeholders}) AND p.sequence > ? {kind_filter} AND (e.parent > ? OR (e.parent = ? AND e.child > ?)) ORDER BY e.parent, e.child LIMIT ?"), - parameters, - }] }).await.map_err(|error| GatewayError::Cell(Box::new(error)))?; - let rows = &result - .output - .first() - .ok_or(GatewayError::MalformedCache)? - .rows; - let mut blobs = Vec::new(); - for row in rows { - let [ - SqlValue::Blob(parent), - SqlValue::Blob(child), - SqlValue::Text(kind), - ] = row.as_slice() - else { - return Err(GatewayError::MalformedCache); - }; - after_parent.clone_from(parent); - after_child.clone_from(child); - let oid = child - .as_slice() - .try_into() - .map_err(|_| GatewayError::MalformedCache)?; - if kind == "blob" { - if include_blobs && visited.insert(oid) { - blobs.push(oid); - } - } else if !visited.contains(&oid) { - pending.insert(oid); - } - } - if !blobs.is_empty() { - self.hydrate_objects(cache, blobs, &mut stats).await?; - } - if rows.len() < MAX_OBJECTS { - break; - } - } - self.hydrate_objects(cache, ids, &mut stats).await?; - } - tracing::debug!( - repository = %hex::encode(self.repository.repository_id()), - objects = stats.objects, - bytes = stats.bytes, - include_blobs, - "prepared reachable Git structure" - ); - Ok(()) - } - - async fn hydrate_objects( - &self, - cache: &Arc, - ids: Vec, - stats: &mut Hydration, - ) -> Result<(), GatewayError> { - let mut missing: BTreeSet<_> = cache.missing_objects(ids).await?.into_iter().collect(); - while !missing.is_empty() { - let selected: Vec<_> = missing.iter().copied().collect(); - let page = self - .repository - .selected_objects(&selected) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - if page.is_empty() { - return Err(GatewayError::MalformedCache); - } - for object in page { - missing.remove(&object.oid); - self.cache_object(cache, object, stats).await?; + return Err(GatewayError::UnreachableWant); } } Ok(()) @@ -447,59 +146,6 @@ impl GitGateway { mod tests { use super::*; - #[test] - fn reachability_stops_at_live_refs_without_scanning_other_history() - -> Result<(), Box> { - use cellule_ltx::rusqlite::{Connection, StatementStatus, params}; - let db = Connection::open_in_memory()?; - db.execute_batch(crate::SCHEMA)?; - let oid = |n: u32| { - let mut oid = [0; 20]; - oid[..4].copy_from_slice(&n.to_be_bytes()); - oid - }; - db.execute_batch("BEGIN")?; - for n in 0..10_000 { - db.execute("INSERT INTO objects (oid, kind, size, digest, storage, body) VALUES (?1, 'blob', 0, zeroblob(32), 'inline', X'')", [oid(n).as_slice()])?; - db.execute( - "INSERT INTO refs (name, oid, version) VALUES (?1, ?2, 1)", - params![format!("refs/tags/{n}"), oid(n).as_slice()], - )?; - if n > 0 { - db.execute( - "INSERT INTO object_edges (parent, child) VALUES (?1, ?2)", - params![oid(n).as_slice(), oid(n - 1).as_slice()], - )?; - } - } - db.execute_batch("COMMIT")?; - // A directly referenced object has thousands of reverse ancestors; - // removing its ref makes the next ancestor the nearest live root. - for direct_ref in [true, false] { - if !direct_ref { - db.execute("DELETE FROM refs WHERE name = 'refs/tags/0'", [])?; - } - let mut query = db.prepare(REACHABLE_WANT)?; - assert_eq!( - query.query_row([oid(0).as_slice()], |row| row.get::<_, i64>(1))?, - 1 - ); - assert!(query.get_status(StatementStatus::VmStep) < 500); - assert_eq!(query.get_status(StatementStatus::FullscanStep), 0); - } - db.execute("DELETE FROM refs", [])?; - let mut query = db.prepare(REACHABLE_WANT)?; - assert_eq!( - query.query_row([oid(0).as_slice()], |row| row.get::<_, i64>(1))?, - 0 - ); - assert_eq!( - query.query_row([oid(10_000).as_slice()], |row| row.get::<_, i64>(1))?, - 0 - ); - Ok(()) - } - fn packet(line: &str) -> Vec { format!("{:04x}{line}", line.len() + 4).into_bytes() } diff --git a/crates/canopy-server/src/git_gateway/hydration.rs b/crates/canopy-server/src/git_gateway/hydration.rs deleted file mode 100644 index 6e646be1..00000000 --- a/crates/canopy-server/src/git_gateway/hydration.rs +++ /dev/null @@ -1,178 +0,0 @@ -use super::*; -use crate::blob::LargeBlobReference; -use std::time::{Duration, Instant}; - -#[derive(Default)] -pub(super) struct Hydration { - pub(super) objects: u64, - pub(super) bytes: u64, - pub(super) body_time: Duration, - pub(super) cache_time: Duration, -} - -impl GitGateway { - pub(super) async fn restore_packs( - &self, - shared: &mut CachedObjects, - ) -> Result<(), GatewayError> { - let mut after = Vec::new(); - loop { - let page = self - .repository - .approved_packs(&after) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - if page.is_empty() { - break; - } - for record in &page { - self.pack_reader.install(&shared.cache, record).await?; - shared.through = shared.through.max(record.covered_through); - } - after = page - .last() - .ok_or(GatewayError::MalformedCache)? - .pack - .sha256 - .to_vec(); - } - Ok(()) - } - - pub(super) async fn hydrate(&self, shared: &mut CachedObjects) -> Result<(), GatewayError> { - let started = Instant::now(); - let cache = &shared.cache; - let _selection = cache.selection.lock().await; - let cursor = &mut shared.through; - let from_sequence = *cursor; - // Bound this refresh even when other writers keep appending objects. - // The read follows the chosen ref snapshot, whose objects are durable. - let high_water = self - .repository - .object_high_water() - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - let mut page_time = Duration::ZERO; - let mut stats = Hydration::default(); - let mut scanned = 0_u64; - while *cursor < high_water.output { - let queried = Instant::now(); - let mut headers = self - .repository - .object_headers(*cursor, &high_water) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - if headers.objects.output.is_empty() { - return Err(GatewayError::MalformedCache); - } - scanned += headers.objects.output.len() as u64; - headers.objects.output = cache.missing_objects(headers.objects.output).await?; - let page = self - .repository - .object_records(headers.objects) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))? - .output; - page_time += queried.elapsed(); - for object in page { - self.cache_object(cache, object, &mut stats).await?; - } - // Failed or cancelled pages retain their previous cursor. Verified - // files can be reused on retry, but no missing body is skipped. - *cursor = headers.through; - } - tracing::debug!( - repository = %hex::encode(self.repository.repository_id()), - objects = stats.objects, - scanned, - from_sequence, - through_sequence = *cursor, - reused = scanned - stats.objects, - bytes = stats.bytes, - cache_bytes = cache.bytes()?, - elapsed_seconds = started.elapsed().as_secs_f64(), - page_seconds = page_time.as_secs_f64(), - body_seconds = stats.body_time.as_secs_f64(), - cache_seconds = stats.cache_time.as_secs_f64(), - "hydrated Git cache" - ); - Ok(()) - } - - pub(super) async fn cache_object( - &self, - cache: &Arc, - object: StoredObject, - stats: &mut Hydration, - ) -> Result, GatewayError> { - let read = Instant::now(); - let body = match object.storage { - ObjectStorage::Inline(body) => body, - ObjectStorage::Packed { pack, size, blake3 } => { - let record = self - .repository - .pack_record(pack) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - if record.approved { - self.pack_reader.install(cache, &record).await?; - return Ok(None); - } - let reader = self - .pack_reader - .native_reader(record, object.oid, size, blake3) - .await?; - stats.body_time += read.elapsed(); - let written = Instant::now(); - cache.store_native_blob(reader).await?; - stats.cache_time += written.elapsed(); - stats.objects += 1; - stats.bytes += size; - return Ok(None); - } - ObjectStorage::Chunked { - upload, - size, - blake3, - } => self - .repository - .chunked_body(object.oid, object.kind, upload, size, blake3) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?, - ObjectStorage::External { - size, - blake3, - sha256, - } => { - let reference = LargeBlobReference { - oid: object.oid, - size, - blake3, - sha256, - }; - let reader = self.large_blobs.read(&reference).await?; - stats.body_time += read.elapsed(); - let written = Instant::now(); - cache.store_blob(reader).await?; - stats.cache_time += written.elapsed(); - stats.objects += 1; - stats.bytes += reference.size; - return Ok(None); - } - }; - stats.body_time += read.elapsed(); - stats.objects += 1; - stats.bytes += body.len() as u64; - let target = (object.kind == ObjectKind::Tag) - .then(|| { - crate::graph::tag_edge(&body) - .map(|(oid, _)| oid) - .filter(|target| target.format() == object.oid.format()) - }) - .flatten(); - let written = Instant::now(); - cache.store_object(object.oid, object.kind, body).await?; - stats.cache_time += written.elapsed(); - Ok(target) - } -} diff --git a/crates/canopy-server/src/git_gateway/maintenance.rs b/crates/canopy-server/src/git_gateway/maintenance.rs deleted file mode 100644 index fd28c88f..00000000 --- a/crates/canopy-server/src/git_gateway/maintenance.rs +++ /dev/null @@ -1,59 +0,0 @@ -use super::*; -use std::sync::atomic::Ordering; - -impl GitGateway { - /// Best-effort maintenance has separate process admission and never owns a - /// user transfer slot. A failed job leaves the previous generation serving. - pub(crate) async fn maintain(&self) -> Result<(), GatewayError> { - static JOBS: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(1); - let Ok(_job) = JOBS.try_acquire() else { - return Ok(()); - }; - let (old, through, generation) = { - let Ok(objects) = self.objects.try_lock() else { - return Ok(()); - }; - let Some(objects) = objects.as_ref() else { - return Ok(()); - }; - if objects.cache.hydrating.load(Ordering::SeqCst) > 0 { - return Ok(()); - } - if objects.cache.loose_objects.load(Ordering::Relaxed) < 1024 - && objects.cache.pack_files.load(Ordering::Relaxed) < 8 - { - return Ok(()); - } - ( - Arc::clone(&objects.cache), - objects.through, - objects.cache.write_generation.load(Ordering::SeqCst), - ) - }; - let started = std::time::Instant::now(); - let next = old - .repacked(self.scratch_root.clone(), self.disk_budget.clone()) - .await?; - // Same lock order as fetch_cache/build_cache. Foreground requests are - // never locked out while pack-objects runs. Changed inventories retry. - let mut refs = self.cache.lock().await; - let mut objects = self.objects.lock().await; - let Some(objects) = objects.as_mut() else { - return Ok(()); - }; - if !Arc::ptr_eq(&objects.cache, &old) - || objects.through != through - || old.hydrating.load(Ordering::SeqCst) > 0 - || old.write_generation.load(Ordering::SeqCst) != generation - { - return Ok(()); - } - tracing::info!(repository = %hex::encode(self.repository.repository_id()), - index_entries = next.indexed_entries(), previous_bytes = old.bytes()?, packed_bytes = next.bytes()?, - elapsed_seconds = started.elapsed().as_secs_f64(), "published background Git repack generation"); - self.pack_reader.replace(&old, Arc::clone(&next)).await; - objects.cache = next; - *refs = None; - Ok(()) - } -} diff --git a/crates/canopy-server/src/git_gateway/mod.rs b/crates/canopy-server/src/git_gateway/mod.rs index ee395661..600bf224 100644 --- a/crates/canopy-server/src/git_gateway/mod.rs +++ b/crates/canopy-server/src/git_gateway/mod.rs @@ -22,28 +22,22 @@ use crate::{ RefUpdate, RepositoryCell, StoredObject, blob::{LargeBlobError, LargeBlobStore}, directory::TokenScope, - git_cache::{CacheError, GitCache}, + git_cache::CacheError, git_http::{GitHttpBackend, GitHttpError, GitHttpRequest, GitHttpResponse}, git_input::{GitInput, InputError, MAX_FETCH_REQUEST_BYTES}, git_objects::GitObjects, lfs::LfsService, - object_batch::MAX_OBJECTS, push::{PushCompletion, PushError}, - refs::RefReadError, }; mod branch_policy; mod candidates; mod discovery; mod fetch; -mod hydration; -mod maintenance; pub mod preflight; mod push; mod ssh; -use hydration::Hydration; - pub use crate::git_objects::ObjectReadError; type CellError = Box; @@ -82,21 +76,9 @@ pub enum GatewayError { Task(#[from] tokio::task::JoinError), } -struct CachedObjects { - cache: Arc, - through: i64, -} - struct CachedRepository { backend: GitHttpBackend, - snapshot: RefSnapshot, -} - -#[derive(PartialEq, Eq)] -struct RefSnapshot { refs: BTreeMap, - head: String, - generation: i64, } /// Serves Git requests from a warm, disposable cache of durable Cell state. @@ -110,8 +92,6 @@ pub struct GitGateway { scratch_root: PathBuf, disk_budget: DiskBudget, native: crate::native_resources::NativeScope, - cache: Mutex>>, - objects: Mutex>, push: Mutex<()>, } @@ -154,8 +134,6 @@ impl GitGateway { scratch_root, disk_budget, native, - cache: Mutex::new(None), - objects: Mutex::new(None), push: Mutex::new(()), } } @@ -250,41 +228,43 @@ impl GitGateway { .receive(request, Some(MAX_FETCH_REQUEST_BYTES), admission) .await?; let request = self.decode(request, Some(MAX_FETCH_REQUEST_BYTES)).await?; + let snapshot = self + .repository + .serving_snapshot(actor) + .await + .map_err(|e| GatewayError::Cell(Box::new(e)))?; let capabilities = request.protocol_v2 && request.method == "GET" && request.path_info == "/repo.git/info/refs" && url::form_urlencoded::parse(request.query.as_bytes()) .eq([("service".into(), "git-upload-pack".into())]); let response = if capabilities { - // Git v2 discovery advertises capabilities, not refs or objects. - // Native Git still owns the wire response and capability policy. - let head = self - .repository - .default_branch(None) + let head = snapshot + .resolve_ref(None) .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; + .map_err(|e| GatewayError::Cell(Box::new(e)))?; let backend = GitHttpBackend::initialize( self.scratch_root.clone(), self.disk_budget.clone(), - &head.output.reference, + &head.reference, self.repository.object_format(), self.native.clone(), ) .await? .with_nonce(self.certificate_nonce().await?); - backend.stream(request, ()).await? - } else if discovery::is_ref_discovery(&request).await? { - if let Some(cached) = self.current_cache().await? { - cached.backend.stream(request, Arc::clone(&cached)).await? - } else { - let backend = self.discovery_cache(self.cell_refs().await?).await?; - backend.stream(request, ()).await? - } + backend.stream(request, snapshot).await? } else { + let discovery = discovery::is_ref_discovery(&request).await?; let fetch = fetch::FetchRequest::read(&request).await?; - let cached = self.fetch_cache(&fetch.wants).await?; - self.prepare_fetch(&cached, fetch).await?; - cached.backend.stream(request, Arc::clone(&cached)).await? + let workspace = snapshot + .ref_workspace(crate::packs::publication::WorkspaceLimits::default()) + .await + .map_err(|e| GatewayError::Cell(Box::new(e)))?; + Self::validate_wants(&workspace, &fetch.wants).await?; + tracing::debug!(discovery,filter=?fetch.filter,generation=workspace.fact().generation, + "prepared certified Git transport"); + let backend = workspace.backend(self.certificate_nonce().await?); + backend.stream(request, workspace.read_owner()).await? }; Ok(GitHttpResponse { status: response.status, @@ -293,33 +273,6 @@ impl GitGateway { }) } - async fn current_cache(&self) -> Result>, GatewayError> { - // Hydration holds this mutex across storage I/O. Discovery must stay - // independent, so inspect only a ready snapshot and release before SQL. - let cached = self.cache.try_lock().ok().and_then(|cache| cache.clone()); - let Some(cached) = cached else { - return Ok(None); - }; - // Ref mutations, deletion/recreation and HEAD changes advance the same - // generation transactionally. Matching it avoids scanning every ref page. - let head = self - .repository - .default_branch(None) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))? - .output; - let current = - cached.snapshot.generation == head.generation && cached.snapshot.head == head.reference; - if current { - tracing::debug!( - repository = %hex::encode(self.repository.repository_id()), - generation = head.generation, - "reused Git ref snapshot" - ); - } - Ok(current.then_some(cached)) - } - async fn receive( &self, request: GitHttpRequest, @@ -373,87 +326,35 @@ impl GitGateway { async fn build_cache( &self, - snapshot: RefSnapshot, - include_blobs: bool, + actor: &str, + names: &[String], ) -> Result { - // Only hydration writes the shared cache, and only from durable Cell - // records. Native pushes/merges write into their private generation. - let mut objects = self.objects.lock().await; - if objects.is_none() { - *objects = Some(CachedObjects { - cache: self.pack_reader.cache().await?, - through: 0, - }); - } - let shared = objects.as_mut().ok_or(GatewayError::MalformedCache)?; - self.restore_packs(shared).await?; - if include_blobs { - self.hydrate(shared).await?; - } else { - // Native ref advertisement only needs tips and peeled tags. Fetch - // hydrates the requested structural graph after validating wants. - self.hydrate_selected( - &shared.cache, - snapshot - .refs - .values() - .filter_map(|state| state.oid) - .collect(), - ) - .await?; + if !valid_ref_names(names) { + return Err(GatewayError::MalformedCache); } - let backend = GitHttpBackend { - cache: GitCache::create_with_objects( - self.scratch_root.clone(), - self.disk_budget.clone(), - &snapshot.head, - self.repository.object_format(), - Some(Arc::clone(&shared.cache)), - self.native.clone(), - ) - .await?, - nonce_seed: self.certificate_nonce().await?, - signers: None, - }; - backend.cache.store_refs(&snapshot.refs).await?; - Ok(CachedRepository { backend, snapshot }) - } - - async fn cell_refs(&self) -> Result { - for _ in 0..3 { - let mut refs = BTreeMap::new(); - let mut after = String::new(); - let mut generation = None; - loop { - let page = match self.repository.refs_page(&after, generation).await { - Ok(page) => page.output, - Err(RefReadError::Changed) => break, - Err(RefReadError::Cell(error)) => { - return Err(GatewayError::Cell(Box::new(error))); - } - }; - generation = Some(page.generation); - let complete = !page.has_more; - for (name, state) in page.refs { - after = name.clone(); - refs.insert(name, state); - } - if complete { - tracing::debug!( - repository = %hex::encode(self.repository.repository_id()), - generation = page.generation, - refs = refs.len(), - "read Git ref snapshot" - ); - return Ok(RefSnapshot { - refs, - head: page.default_branch, - generation: page.generation, - }); + let snapshot = self + .repository + .serving_snapshot(ReadIdentity::Account(actor)) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))?; + let mut refs = BTreeMap::new(); + for page in ref_pages(names, 128, 256 << 10) { + for resolved in snapshot + .resolve_refs(page) + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))? + { + if let Some(state) = resolved.state { + refs.insert(resolved.reference, state); } } } - Err(GatewayError::RefSnapshotBusy) + let backend = snapshot + .native_base() + .await + .map_err(|error| GatewayError::Cell(Box::new(error)))? + .with_nonce(self.certificate_nonce().await?); + Ok(CachedRepository { backend, refs }) } async fn persist_objects( @@ -474,12 +375,6 @@ impl GitGateway { // re-reading old history; the final Cell transaction still verifies every new tip. let excluded = before.values().filter_map(|state| state.oid).collect(); let started = std::time::Instant::now(); - let initial_high_water = self - .repository - .object_high_water() - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))? - .output; let mut sources = backend.cache.pack_sources().await?; let mut archive = None; let mut packed_ids = None; @@ -490,7 +385,6 @@ impl GitGateway { pack: self.pack_reader.upload(pack).await?, index: self.pack_reader.upload(index).await?, approved: false, - covered_through: 0, }; self.repository .register_pack(new_identity()?, &record) @@ -634,42 +528,6 @@ impl GitGateway { .await .map_err(|error| GatewayError::Cell(Box::new(error)))?; } - let mut shared = self.objects.lock().await; - if let Some(shared) = shared.as_mut() { - let count = verified.len(); - match shared - .cache - .retain_verified_packs(Arc::clone(&backend.cache), verified) - .await - { - Ok(retained) if retained == count && shared.through == initial_high_water => { - if let Some(record) = &archive { - shared.cache.mark_durable_pack(record.pack.sha256); - } - shared.through = self - .repository - .object_high_water() - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))? - .output; - tracing::info!( - objects = retained, - through = shared.through, - "retained verified receive pack for immediate fetch" - ); - } - Ok(retained) => { - if retained == count - && let Some(record) = &archive - { - shared.cache.mark_durable_pack(record.pack.sha256); - } - } - Err(error) => { - tracing::warn!(error = ?error, "receive pack cache reuse skipped; durable hydration remains available") - } - } - } tracing::info!( elapsed_seconds = started.elapsed().as_secs_f64(), "persisted Git objects" @@ -694,79 +552,80 @@ fn with_push_id(mut response: GitHttpResponse, id: [u8; 16]) -> GitHttpResponse response } -async fn git_output( - git_dir: &Path, - args: &[&str], - native: &crate::native_resources::NativeScope, -) -> Result, GatewayError> { - use crate::git_http::{GitProcess, WORKER_DEADLINE, read_bounded}; - use tokio::io::AsyncReadExt; - let mut command = crate::native_git::command(git_dir)?; - command - .arg("--git-dir") - .arg(git_dir) - .args(args) - .stdin(std::process::Stdio::null()) - .stdout(std::process::Stdio::piped()) - .stderr(std::process::Stdio::piped()); - let mut process = GitProcess::spawn( - command, - (), - native.try_admit(crate::native_resources::NativeWork::Read)?, - )?; - let mut stdout = process - .child - .stdout - .take() - .ok_or(GatewayError::MalformedCache)?; - let stderr = process - .child - .stderr - .take() - .ok_or(GatewayError::MalformedCache)?; - let run = async { - let mut bytes = Vec::new(); - let read_stdout = async { - stdout.read_to_end(&mut bytes).await?; - Ok::<_, GitHttpError>(()) - }; - let ((), stderr) = tokio::try_join!(read_stdout, read_bounded(stderr, 64 << 10))?; - let status = process.wait().await?; - if !status.success() { - return Err(GatewayError::Git( - String::from_utf8_lossy(&stderr).into_owned(), - )); +fn valid_ref_names(names: &[String]) -> bool { + names.len() <= crate::refs::MAX_UPDATES + && names.windows(2).all(|p| p[0] < p[1]) + && names.iter().all(|name| { + name.len() <= crate::packs::ref_state::MAX_NAME_BYTES + && crate::refs::valid_ref_name(name) + }) +} + +fn ref_pages(names: &[String], count: usize, bytes: usize) -> impl Iterator { + let mut at = 0; + std::iter::from_fn(move || { + if at == names.len() { + return None; + } + let start = at; + let mut used = 0; + while at < names.len() && at - start < count { + let charge = names[at].len() + 1; + if charge > bytes - used { + break; + } + used += charge; + at += 1; } - Ok(bytes) - }; - tokio::time::timeout(WORKER_DEADLINE, run) - .await - .map_err(|_| GitHttpError::Timeout)? + // Callers validate names against MAX_NAME_BYTES, so one always fits. + Some(&names[start..at]) + }) } +// Exact requested names only. for-each-ref patterns can scan entire subtrees +// when an absent requested name prefixes existing refs; cat-file resolves each +// validated literal ref independently and reports missing names in order. async fn git_refs( - git_dir: &Path, - native: &crate::native_resources::NativeScope, + backend: &GitHttpBackend, + names: &[String], ) -> Result, GatewayError> { - let listing = git_output( - git_dir, - &["for-each-ref", "--format=%(refname)%00%(objectname)"], - native, - ) - .await?; + if !valid_ref_names(names) { + return Err(GatewayError::MalformedCache); + } let mut refs = BTreeMap::new(); - for line in listing - .split(|byte| *byte == b'\n') - .filter(|line| !line.is_empty()) - { - let Some(separator) = line.iter().position(|byte| *byte == 0) else { + for page in ref_pages(names, 32, 64 << 10) { + let mut input = page.join("\n").into_bytes(); + input.push(b'\n'); + let output = candidates::run( + backend, + &["cat-file", "--batch-check=%(objectname)"], + &input, + &[], + ) + .await?; + if !output.status.success() { + return Err(output.error()); + } + let text = std::str::from_utf8(&output.stdout).map_err(|_| GatewayError::MalformedCache)?; + let lines = text + .strip_suffix('\n') + .ok_or(GatewayError::MalformedCache)? + .split('\n') + .collect::>(); + if lines.len() != page.len() { return Err(GatewayError::MalformedCache); - }; - let name = - std::str::from_utf8(&line[..separator]).map_err(|_| GatewayError::MalformedCache)?; - let oid = std::str::from_utf8(&line[separator + 1..]) - .map_err(|_| GatewayError::MalformedCache)?; - refs.insert(name.to_owned(), parse_oid(oid)?); + } + for (name, line) in page.iter().zip(lines) { + if line == format!("{name} missing") { + continue; + } + let id = crate::ObjectId::from_hex(line.as_bytes()) + .map_err(|_| GatewayError::MalformedCache)?; + if id.is_zero() || id.format() != backend.cache.object_format { + return Err(GatewayError::MalformedCache); + } + refs.insert(name.clone(), id); + } } Ok(refs) } @@ -823,3 +682,95 @@ fn new_identity() -> Result { expires_at_ms: now_ms + 60_000, }) } + +#[cfg(test)] +mod native_refs_tests { + use super::*; + #[tokio::test] + async fn native_ref_reads_resolve_only_exact_requested_names_in_both_formats() + -> Result<(), Box> { + for format in [crate::ObjectFormat::Sha1, crate::ObjectFormat::Sha256] { + let root = tempfile::TempDir::new()?; + let backend = GitHttpBackend::initialize( + root.path().to_owned(), + DiskBudget::new(8 << 20), + "refs/heads/main", + format, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let body = b"native ref lookup"; + let id = crate::object_id(format, ObjectKind::Blob, body); + backend + .cache + .store_object(id, ObjectKind::Blob, body.to_vec()) + .await?; + backend + .cache + .store_refs(&BTreeMap::from([ + ( + "refs/heads/main".into(), + RefExpectation { + oid: Some(id), + version: 1, + }, + ), + ( + "refs/heads/absent/child".into(), + RefExpectation { + oid: Some(id), + version: 1, + }, + ), + ( + "refs/tags/tag".into(), + RefExpectation { + oid: Some(id), + version: 1, + }, + ), + ])) + .await?; + let names = vec![ + "refs/heads/absent".into(), + "refs/heads/main".into(), + "refs/tags/missing".into(), + "refs/tags/tag".into(), + ]; + assert_eq!( + git_refs(&backend, &names).await?, + BTreeMap::from([("refs/heads/main".into(), id), ("refs/tags/tag".into(), id)]) + ); + assert!(matches!( + git_refs(&backend, &["HEAD".into()]).await, + Err(GatewayError::MalformedCache) + )); + } + Ok(()) + } + #[test] + fn ref_pages_bound_names_and_bytes_and_reject_non_literal_input() { + let long = format!( + "refs/heads/{}", + "x".repeat(crate::packs::ref_state::MAX_NAME_BYTES - 11) + ); + let names = vec![long, "refs/tags/a".into(), "refs/tags/b".into()]; + assert!(valid_ref_names(&names)); + let pages: Vec<_> = ref_pages(&names, 32, 64 << 10).collect(); + assert_eq!(pages.iter().map(|p| p.len()).collect::>(), [1, 2]); + assert_eq!(pages.concat(), names); + for name in [ + "HEAD", + "refs/heads/a^", + "refs/heads/a\n", + "refs/heads/a:foo", + ] { + assert!(!valid_ref_names(&[name.into()])); + } + assert!(!valid_ref_names(&[format!( + "refs/heads/{}", + "x".repeat(65_536) + )])); + } +} diff --git a/crates/canopy-server/src/git_gateway/push.rs b/crates/canopy-server/src/git_gateway/push.rs index 97816093..0fca0d64 100644 --- a/crates/canopy-server/src/git_gateway/push.rs +++ b/crates/canopy-server/src/git_gateway/push.rs @@ -24,10 +24,11 @@ impl GitGateway { .ok_or(GatewayError::MalformedCache)?; return Ok::<_, GatewayError>((response, None, None)); } - let cached = self.build_cache(self.cell_refs().await?, true).await?; + let names = commands.names(); + let cached = self.build_cache(actor, &names).await?; self.install_branch_policy(&cached, &commands).await?; let signers = self.install_certificate_policy(&cached, &commands, actor).await?; - let before = cached.snapshot.refs.clone(); + let before = cached.refs.clone(); let backend = signers.map_or_else( || cached.backend.clone(), |path| cached.backend.with_signers(path), @@ -37,7 +38,7 @@ impl GitGateway { // Git may accept some refs and reject others unless atomic was requested. // Publish its actual changes before returning any successful per-ref report. let plan = if response.status == 200 { - let after = git_refs(&cached.backend.git_dir(), &cached.backend.cache.native).await?; + let after = git_refs(&cached.backend, &names).await?; let plan = diff_refs(&before, &after, actor); if plan.updates.is_empty() { None diff --git a/crates/canopy-server/src/git_gateway/ssh.rs b/crates/canopy-server/src/git_gateway/ssh.rs index e7160455..217b9c30 100644 --- a/crates/canopy-server/src/git_gateway/ssh.rs +++ b/crates/canopy-server/src/git_gateway/ssh.rs @@ -20,14 +20,20 @@ impl GitGateway { if self.access_level(actor).await?.is_none() { return Err(GatewayError::Unauthorized); } - // Advertisements need structure and ref/tag targets, not ordinary blobs. - // Retain one ref snapshot, then hydrate each request before forwarding - // its wants; native Git can begin object traversal as soon as it reads them. - let cached = self.fetch_cache(&BTreeSet::new()).await?; - let mut command = cached.backend.transport_command()?; + let snapshot = self + .repository + .serving_snapshot(ReadIdentity::Account(actor)) + .await + .map_err(|e| GatewayError::Cell(Box::new(e)))?; + let workspace = snapshot + .ref_workspace(crate::packs::publication::WorkspaceLimits::default()) + .await + .map_err(|e| GatewayError::Cell(Box::new(e)))?; + let backend = workspace.backend(self.certificate_nonce().await?); + let mut command = backend.transport_command()?; command .arg("upload-pack") - .arg(cached.backend.git_dir()) + .arg(backend.git_dir()) .stdin(Stdio::piped()) .stdout(Stdio::piped()) .stderr(Stdio::piped()); @@ -36,9 +42,8 @@ impl GitGateway { } let mut process = GitProcess::spawn( command, - (Arc::clone(&cached), admission), - cached - .backend + (workspace.read_owner(), admission), + backend .cache .native .try_admit(crate::native_resources::NativeWork::Pack)?, @@ -68,9 +73,7 @@ impl GitGateway { }; remaining -= group.len(); let request = fetch::FetchRequest::parse(&group)?; - self.validate_wants(&cached.snapshot, &request.wants) - .await?; - self.prepare_fetch(&cached, request).await?; + Self::validate_wants(&workspace, &request.wants).await?; stdin.write_all(&group).await?; stdin.flush().await?; if !protocol_v2 { diff --git a/crates/canopy-server/src/git_http/capture.rs b/crates/canopy-server/src/git_http/capture.rs index 28423062..1bc10afb 100644 --- a/crates/canopy-server/src/git_http/capture.rs +++ b/crates/canopy-server/src/git_http/capture.rs @@ -41,6 +41,7 @@ struct CapturePin { // captured file immutable while hash/upload background jobs retain it. _fence: File, cache: Arc, + _owner: crate::git_objects::ReadOwner, } impl Drop for CapturePin { fn drop(&mut self) { @@ -90,9 +91,10 @@ impl GitHttpBackend { return Err(NativeCaptureError::Limit); } let _selection = self.cache.selection.lock().await; - self.cache.reconcile().await?; + self.cache.reconcile_owned(context.physical_owner()).await?; let cache = Arc::clone(&self.cache); let format = context.format(); + let owner = context.physical_owner(); let claim = cache .native .try_admit(crate::native_resources::NativeWork::Read)?; @@ -105,6 +107,7 @@ impl GitHttpBackend { let pin = Arc::new(CapturePin { cache, _fence: fence, + _owner: owner, }); for entry in std::fs::read_dir(pin.cache.git_dir().join("objects"))? { let entry = entry?; @@ -234,6 +237,7 @@ mod tests { _pin: Arc::new(CapturePin { _fence: fence, cache: backend.cache.clone(), + _owner: Arc::new(()), }), }); let retained = captured.clone(); @@ -312,6 +316,7 @@ mod tests { _pin: Arc::new(CapturePin { _fence: fence, cache: Arc::clone(&backend.cache), + _owner: Arc::new(()), }), }); let (ready_tx, ready_rx) = std::sync::mpsc::channel(); diff --git a/crates/canopy-server/src/git_http/mod.rs b/crates/canopy-server/src/git_http/mod.rs index 43f2ddcf..6e79505f 100644 --- a/crates/canopy-server/src/git_http/mod.rs +++ b/crates/canopy-server/src/git_http/mod.rs @@ -38,6 +38,8 @@ pub(crate) const WORKER_DEADLINE: Duration = Duration::from_secs(3600); #[derive(Debug, thiserror::Error)] pub enum GitHttpError { + #[error("native receive has no live staging custody")] + Staging(#[from] crate::packs::publication::StagingError), #[error("Git cache failed")] Cache(#[from] CacheError), #[error("Git process I/O failed")] @@ -133,8 +135,13 @@ impl GitHttpBackend { /// this API, then capture and verify inputs before durable publication. pub async fn run_native_receive( &self, + context: &crate::packs::publication::StagingContext, request: GitHttpRequest, ) -> Result { + context.ensure_live()?; + if context.format() != self.cache.object_format || !request.authenticated { + return Err(GitHttpError::Interrupted); + } if request.method != "POST" || request.path_info != "/repo.git/git-receive-pack" || !request.query.is_empty() @@ -143,13 +150,27 @@ impl GitHttpBackend { } let mut command = self.transport_command()?; command.args(["-c", "receive.unpackLimit=0"]); - let response = self.stream_command(request, (), command).await?; - self.collect(response).await + let response = self + .stream_command(request, context.physical_owner(), command) + .await?; + let response = self + .collect_owned(response, context.physical_owner()) + .await?; + context.ensure_live()?; + Ok(response) } async fn collect( &self, response: GitHttpResponse, + ) -> Result { + self.collect_owned(response, Arc::new(())).await + } + + async fn collect_owned( + &self, + response: GitHttpResponse, + owner: crate::git_objects::ReadOwner, ) -> Result { let GitHttpResponse { status, @@ -164,7 +185,7 @@ impl GitHttpBackend { } bytes.extend_from_slice(&chunk); } - self.cache.reconcile().await?; + self.cache.reconcile_owned(owner).await?; Ok(GitHttpResponse { status, headers, diff --git a/crates/canopy-server/src/git_objects/mod.rs b/crates/canopy-server/src/git_objects/mod.rs index 7e952953..fe593e9f 100644 --- a/crates/canopy-server/src/git_objects/mod.rs +++ b/crates/canopy-server/src/git_objects/mod.rs @@ -33,8 +33,11 @@ pub enum ObjectReadError { Task(#[from] tokio::task::JoinError), } +pub(crate) type ReadOwner = std::sync::Arc; + struct Process { - worker: crate::native_git::process::GitProcess<()>, + owner: ReadOwner, + worker: crate::native_git::process::GitProcess, output: BufReader, stderr: AbortOnDropHandle, io::Error>>, } @@ -44,6 +47,15 @@ impl Process { git_dir: &Path, args: &[&str], native: &crate::native_resources::NativeScope, + ) -> Result<(Self, ChildStdin), ObjectReadError> { + Self::start_owned(git_dir, args, native, std::sync::Arc::new(())) + } + + fn start_owned( + git_dir: &Path, + args: &[&str], + native: &crate::native_resources::NativeScope, + owner: ReadOwner, ) -> Result<(Self, ChildStdin), ObjectReadError> { let mut command = crate::native_git::command(git_dir)?; command @@ -55,7 +67,7 @@ impl Process { .stderr(Stdio::piped()); let mut child = crate::native_git::process::GitProcess::spawn( command, - (), + std::sync::Arc::clone(&owner), native.try_admit(crate::native_resources::NativeWork::Read)?, )?; let input = child.child.stdin.take().ok_or(ObjectReadError::Malformed)?; @@ -84,6 +96,7 @@ impl Process { })); Ok(( Self { + owner, worker: child, output: BufReader::new(output), stderr, @@ -112,6 +125,7 @@ pub(crate) struct GitObjectWalk { } impl GitObjectWalk { + #[cfg(test)] pub(crate) fn missing( git_dir: &Path, included: Vec, @@ -236,7 +250,18 @@ impl GitObjects { git_dir: &Path, native: &crate::native_resources::NativeScope, ) -> Result { - let (batch, requests) = Process::start(git_dir, &["cat-file", "--batch"], native)?; + Self::batch_owned(git_dir, native, std::sync::Arc::new(())) + } + + /// Retain physical generation/cache admission through native descendants + /// and deferred reaping, including cancellation of the calling worker. + pub(crate) fn batch_owned( + git_dir: &Path, + native: &crate::native_resources::NativeScope, + owner: ReadOwner, + ) -> Result { + let (batch, requests) = + Process::start_owned(git_dir, &["cat-file", "--batch"], native, owner)?; Ok(Self { walk: None, inventory: None, @@ -246,6 +271,34 @@ impl GitObjects { }) } + /// Bounded canonical body read. A canceled, rejected or corrupt response + /// poisons the batch; only a fully verified frame permits reuse. + pub(crate) async fn read_verified( + &mut self, + expected: crate::packs::metadata::CanonicalObject, + limit: usize, + ) -> Result, ObjectReadError> { + if self.inspection_failed { + return Err(ObjectReadError::Malformed); + } + if expected.size > limit as u64 { + return Err(ObjectReadError::TooLarge); + } + self.inspection_failed = true; + let owner = std::sync::Arc::clone(&self.batch.owner); + let object = timeout(IO_TIMEOUT, async { + self.requests + .write_all(format!("{}\n", hex::encode(expected.oid)).as_bytes()) + .await?; + open_object(&mut self.batch.output, expected.oid).await + }) + .await + .map_err(|_| ObjectReadError::Timeout)??; + let body = object.body_verified(expected, limit, owner).await?; + self.inspection_failed = false; + Ok(body) + } + /// Streams canonical hashing and typed structural extraction. Sink writes /// are private preparation; discard them if this returns an error or is /// canceled. Pack binding and graph closure remain verifier obligations. @@ -455,6 +508,39 @@ impl GitObject<'_, R> { Ok(()) } + async fn body_verified( + mut self, + expected: crate::packs::metadata::CanonicalObject, + limit: usize, + owner: ReadOwner, + ) -> Result, ObjectReadError> { + if self.oid != expected.oid || self.kind != expected.kind || self.size != expected.size { + return Err(ObjectReadError::Malformed); + } + if self.size > limit as u64 || self.size > isize::MAX as u64 { + return Err(ObjectReadError::TooLarge); + } + let mut body = Vec::new(); + body.try_reserve_exact(self.size as usize) + .map_err(ObjectReadError::Allocation)?; + body.resize(self.size as usize, 0); + timeout(IO_TIMEOUT, self.reader.read_exact(&mut body)) + .await + .map_err(|_| ObjectReadError::Timeout)??; + self.finish().await?; + // Do not drop a detached hash job's physical owner at an observer timeout. + tokio::task::spawn_blocking(move || { + let _owner = owner; + if object_id(expected.oid.format(), expected.kind, &body) != expected.oid + || blake3::hash(&body).as_bytes() != &expected.digest + { + return Err(ObjectReadError::Malformed); + } + Ok(body) + }) + .await? + } + pub(crate) async fn body(mut self) -> Result<(ObjectKind, Vec), ObjectReadError> { let limit = if self.kind == ObjectKind::Blob { INLINE_OBJECT_LIMIT diff --git a/crates/canopy-server/src/git_objects/tests.rs b/crates/canopy-server/src/git_objects/tests.rs index b8726384..52b85028 100644 --- a/crates/canopy-server/src/git_objects/tests.rs +++ b/crates/canopy-server/src/git_objects/tests.rs @@ -342,3 +342,133 @@ async fn streamed_inspection_rejects_hash_mismatch_partial_bodies_bad_separators } Ok(()) } + +#[tokio::test] +async fn verified_body_checks_catalog_fingerprint_size_kind_and_limit_for_both_formats() +-> TestResult { + for format in [crate::ObjectFormat::Sha1, crate::ObjectFormat::Sha256] { + let body = b"bounded canonical body"; + let expected = crate::packs::metadata::CanonicalObject { + oid: object_id(format, ObjectKind::Blob, body), + kind: ObjectKind::Blob, + size: body.len() as u64, + digest: *blake3::hash(body).as_bytes(), + }; + let frame = format!("{} blob {}\n", hex::encode(expected.oid), expected.size) + .into_bytes() + .into_iter() + .chain(body.iter().copied()) + .chain(*b"\n") + .collect::>(); + let mut input = frame.as_slice(); + let object = open_object(&mut input, expected.oid).await?; + assert_eq!( + object + .body_verified(expected, body.len(), std::sync::Arc::new(())) + .await?, + body + ); + for variant in 0..4 { + let mut metadata = expected; + let mut limit = body.len(); + match variant { + 0 => metadata.digest[0] ^= 1, + 1 => metadata.kind = ObjectKind::Tree, + 2 => metadata.size += 1, + _ => limit -= 1, + } + let mut input = frame.as_slice(); + let object = open_object(&mut input, expected.oid).await?; + assert!( + object + .body_verified(metadata, limit, std::sync::Arc::new(())) + .await + .is_err() + ); + } + } + Ok(()) +} + +#[tokio::test] +async fn verified_batch_reuses_only_complete_verified_frames_and_poison_refuses_finish() +-> TestResult { + let directory = fixture().await?; + let blob = oid(directory.path(), "HEAD:file-0").await?; + let expected = crate::packs::metadata::CanonicalObject { + oid: blob, + kind: ObjectKind::Blob, + size: 3, + digest: *blake3::hash(b"0\0\n").as_bytes(), + }; + let resources = crate::native_resources::NativeResources::default(); + let scope = resources.scope(crate::native_resources::NativeClass::Foreground); + let mut reader = GitObjects::batch(&directory.path().join(".git"), &scope)?; + assert!(matches!( + reader.read_verified(expected, 2).await, + Err(ObjectReadError::TooLarge) + )); + assert_eq!(reader.read_verified(expected, 3).await?, b"0\0\n"); + assert_eq!(reader.read_verified(expected, 3).await?, b"0\0\n"); + reader.finish().await?; + let mut reader = GitObjects::batch(&directory.path().join(".git"), &scope)?; + let mut corrupt = expected; + corrupt.digest[0] ^= 1; + assert!(matches!( + reader.read_verified(corrupt, 3).await, + Err(ObjectReadError::Malformed) + )); + assert!(matches!( + reader.read_verified(expected, 3).await, + Err(ObjectReadError::Malformed) + )); + assert!(matches!( + reader.finish().await, + Err(ObjectReadError::Malformed) + )); + Ok(()) +} + +#[cfg(unix)] +#[tokio::test] +async fn owned_batch_cancellation_releases_owner_only_after_native_reaping() -> TestResult { + use std::sync::{ + Arc, + atomic::{AtomicU32, Ordering}, + }; + struct Owner { + pid: Arc, + released: tokio::sync::oneshot::Sender, + } + impl Drop for Owner { + fn drop(&mut self) { + let pid = self.pid.load(Ordering::Acquire); + // SAFETY: signal zero only queries a PID assigned by this fixture. + let gone = unsafe { libc::kill(pid as i32, 0) } == -1 + && io::Error::last_os_error().raw_os_error() == Some(libc::ESRCH); + let (unused, _) = tokio::sync::oneshot::channel(); + let released = std::mem::replace(&mut self.released, unused); + let _ = released.send(gone); + } + } + let directory = fixture().await?; + let pid = Arc::new(AtomicU32::new(0)); + let (released, observed) = tokio::sync::oneshot::channel(); + let owner = Arc::new(Owner { + pid: Arc::clone(&pid), + released, + }); + let reader = GitObjects::batch_owned( + &directory.path().join(".git"), + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + owner, + )?; + pid.store( + reader.batch.worker.child.id().ok_or("native PID")?, + Ordering::Release, + ); + drop(reader); + assert!(timeout(Duration::from_secs(5), observed).await??); + Ok(()) +} diff --git a/crates/canopy-server/src/git_read/browse.rs b/crates/canopy-server/src/git_read/browse.rs index 61ae7e21..ebe0722f 100644 --- a/crates/canopy-server/src/git_read/browse.rs +++ b/crates/canopy-server/src/git_read/browse.rs @@ -1,6 +1,7 @@ use super::*; use crate::ReadIdentity; -use crate::refs::{RefReadError, valid_ref_name}; +use crate::packs::ref_state::MAX_NAME_BYTES; +use crate::refs::valid_ref_name; const PAGE: usize = 32; @@ -65,21 +66,22 @@ impl Reader { generation: Option, ) -> Result { let actor = actor.into(); - if (!after.is_empty() && (!valid_ref_name(after) || generation.is_none())) + if (!after.is_empty() + && (after.len() > MAX_NAME_BYTES || !valid_ref_name(after) || generation.is_none())) || generation.is_some_and(|n| n < 0) { return Err(ReadError::Invalid); } self.member(actor).await?; - let page = self - .repository - .refs_page(after, generation) - .await - .map_err(|error| match error { - RefReadError::Changed => ReadError::Changed, - RefReadError::Cell(error) => ReadError::Cell(error), - })? - .output; + let snapshot = self.repository.serving_snapshot(actor).await?; + let page = + snapshot + .refs_page(after, generation, true) + .await + .map_err(|error| match error { + crate::packs::publication::ServingReadError::Changed => ReadError::Changed, + error => ReadError::Serving(error), + })?; let next_after = page .has_more .then(|| page.refs.last().map(|(name, _)| name.clone())) @@ -96,62 +98,30 @@ impl Reader { reference: Option<&str>, ) -> Result { let actor = actor.into(); - if reference.is_some_and(|name| !valid_ref_name(name)) { + if reference.is_some_and(|name| name.len() > MAX_NAME_BYTES || !valid_ref_name(name)) { return Err(ReadError::Invalid); } self.member(actor).await?; - // Resolve default HEAD and its tip in one observation. The returned OID - // pins subsequent reads even if a writer moves the ref during browsing. - let result = self.repository.sql.query(None, SqlBatch {statements: vec![SqlStatement { - sql: "SELECT g.generation, coalesce(?1,g.default_branch), r.oid, r.version FROM ref_generation g LEFT JOIN refs r ON r.name = coalesce(?1,g.default_branch) WHERE g.singleton = 1".into(), - parameters: vec![reference.map_or(SqlValue::Null, |value| SqlValue::Text(value.into()))], - }]}).await?; - let row = result - .output - .first() - .and_then(|set| set.rows.first()) - .ok_or(ReadError::Malformed)?; - let [ - SqlValue::Integer(generation), - SqlValue::Text(name), - value, - version, - ] = row.as_slice() - else { - return Err(ReadError::Malformed); - }; - let oid = match value { - SqlValue::Null => None, - SqlValue::Blob(bytes) if matches!(bytes.len(), 20 | 32) => Some(hex::encode(bytes)), - _ => return Err(ReadError::Malformed), - }; - let version = match version { - SqlValue::Null => None, - SqlValue::Integer(n) => Some(*n), - _ => return Err(ReadError::Malformed), - }; + let snapshot = self.repository.serving_snapshot(actor).await?; + let resolved = snapshot.resolve_ref(reference).await?; + let oid = resolved + .state + .as_ref() + .and_then(|state| state.oid) + .map(hex::encode); + let version = resolved.state.as_ref().map(|state| state.version); self.member(actor).await?; - Ok( - serde_json::json!({"reference":name,"oid":oid,"version":version,"generation":generation}), - ) + Ok(serde_json::json!({ + "reference":resolved.reference, "oid":oid, + "version":version, "generation":resolved.generation, + })) } async fn commit(&mut self, mut target: Oid) -> Result { // Only certified objects are browseable. Staged push bytes do not become // visible through a guessed OID before their complete graph is verified. for _ in 0..16 { - let rows = self.repository.sql.query(None,SqlBatch {statements:vec![SqlStatement { - sql:"SELECT o.kind FROM objects o JOIN object_closure c ON c.oid = o.oid WHERE o.oid = ?1".into(), parameters:vec![SqlValue::Blob(target.to_vec())], - }]}).await?; - let Some([SqlValue::Text(kind)]) = rows - .output - .first() - .and_then(|s| s.rows.first()) - .map(Vec::as_slice) - else { - return Err(ReadError::Missing); - }; - match kind.as_str() { - "commit" => { + match self.header(target).await?.object.kind { + ObjectKind::Commit => { let body = self.body(target, ObjectKind::Commit).await?; let admission = Arc::clone(&self.admission); return tokio::task::spawn_blocking(move || { @@ -160,7 +130,7 @@ impl Reader { }) .await?; } - "tag" => { + ObjectKind::Tag => { let body = self.body(target, ObjectKind::Tag).await?; let first = body .split(|b| *b == b'\n') @@ -187,6 +157,7 @@ impl Reader { ) -> Result { let actor = actor.into(); self.member(actor).await?; + self.bind(actor).await?; let path = if encoded_path.is_empty() { Vec::new() } else { @@ -243,6 +214,7 @@ impl Reader { ) -> Result { let actor = actor.into(); self.member(actor).await?; + self.bind(actor).await?; let path = path(encoded_path)?; let commit = self.commit(oid(revision)?).await?; let entry = self @@ -282,6 +254,7 @@ impl Reader { ) -> Result { let actor = actor.into(); self.member(actor).await?; + self.bind(actor).await?; let mut target = Some(oid(revision)?); let mut commits = Vec::new(); while let Some(current) = target { diff --git a/crates/canopy-server/src/git_read/graph.rs b/crates/canopy-server/src/git_read/graph.rs index c4f546ff..1502209b 100644 --- a/crates/canopy-server/src/git_read/graph.rs +++ b/crates/canopy-server/src/git_read/graph.rs @@ -3,13 +3,23 @@ use std::collections::{HashMap, HashSet}; const MAX_COMMITS: usize = 100_000; const MAX_EDGES: usize = 250_000; -const GROUP: usize = 128; -const PAGE: usize = 512; +const GROUP: usize = crate::packs::publication::MAX_EDGE_PARENTS; type Graph = HashMap>; impl Reader { pub(super) async fn merge_base(&self, base: Oid, source: Oid) -> Result { + if base.format() != self.repository.object_format() + || source.format() != self.repository.object_format() + { + return Err(ReadError::Invalid); + } + if base.is_zero() || source.is_zero() { + return Err(ReadError::Missing); + } if base == source { + if self.header(base).await?.object.kind != ObjectKind::Commit { + return Err(ReadError::Malformed); + } return Ok(base); } let mut graph = Graph::new(); @@ -23,35 +33,20 @@ impl Reader { for oid in &group { graph.insert(*oid, Vec::new()); } - let mut cursor: Option<(Oid, Oid)> = None; + let mut ids = group; + ids.sort_unstable(); + let mut cursor = None; loop { - let placeholders = (3..group.len() + 3) - .map(|n| format!("?{n}")) - .collect::>() - .join(","); - let mut parameters = vec![ - SqlValue::Blob(cursor.map_or_else(Vec::new, |(child, _)| child.to_vec())), - SqlValue::Blob(cursor.map_or_else(Vec::new, |(_, parent)| parent.to_vec())), - ]; - parameters.extend(group.iter().map(|oid| SqlValue::Blob(oid.to_vec()))); - // Parent rows come only from verified commit certificates. Keyset - // paging includes every parent even for unusually wide merges. - let result = self.repository.sql.query(None,SqlBatch { statements:vec![SqlStatement { - sql:format!("SELECT child, parent FROM commit_parents WHERE child IN ({placeholders}) AND (child > ?1 OR (child = ?1 AND parent > ?2)) ORDER BY child, parent LIMIT {PAGE}"),parameters, - }] }).await?; - let rows = &result.output.first().ok_or(ReadError::Malformed)?.rows; - for row in rows { - let [SqlValue::Blob(child), SqlValue::Blob(parent)] = row.as_slice() else { + let page = self.snapshot()?.edges_page(&ids, cursor).await?; + for (_, header) in page.headers { + if header.ok_or(ReadError::Missing)?.object.kind != ObjectKind::Commit { return Err(ReadError::Malformed); - }; - let child: Oid = child - .as_slice() - .try_into() - .map_err(|_| ReadError::Malformed)?; - let parent: Oid = parent - .as_slice() - .try_into() - .map_err(|_| ReadError::Malformed)?; + } + } + for (child, edge) in page.edges { + if edge.expected_kind != ObjectKind::Commit { + continue; + } edges += 1; if edges > MAX_EDGES { return Err(ReadError::TooLarge); @@ -59,18 +54,19 @@ impl Reader { graph .get_mut(&child) .ok_or(ReadError::Malformed)? - .push(parent); - cursor = Some((child, parent)); - if discovered.insert(parent) { + .push(edge.child); + if discovered.insert(edge.child) { if discovered.len() > MAX_COMMITS { return Err(ReadError::TooLarge); } - pending.push(parent); + pending.push(edge.child); } } - if rows.len() < PAGE { - break; + let Some(next) = page.next_after else { break }; + if cursor.is_some_and(|old| old >= next) { + return Err(ReadError::Malformed); } + cursor = Some(next); } } let admission = Arc::clone(&self.admission); diff --git a/crates/canopy-server/src/git_read/mod.rs b/crates/canopy-server/src/git_read/mod.rs index 2106d4a1..d952f5c4 100644 --- a/crates/canopy-server/src/git_read/mod.rs +++ b/crates/canopy-server/src/git_read/mod.rs @@ -1,4 +1,4 @@ -//! Bounded repository browsing and PR comparison over verified Cell Git objects. +//! Bounded repository browsing and comparison over certified immutable Git objects. use crate::ReadIdentity; @@ -13,10 +13,7 @@ use crate::{ pulls::{PullRevision, parse_oid}, }; use base64::{Engine, engine::general_purpose::URL_SAFE_NO_PAD}; -use cellule_runtime::{ - InvocationError, primitives::sql::SqlBatch, primitives::sql::SqlResultSet, - primitives::sql::SqlStatement, primitives::sql::SqlValue, -}; +use cellule_runtime::{InvocationError, primitives::sql::SqlResultSet}; use serde::{Deserialize, Serialize}; use std::{collections::BTreeMap, sync::Arc}; @@ -50,6 +47,10 @@ pub(crate) enum ReadError { Cell(#[from] InvocationError>), #[error("Git read worker failed")] Task(#[from] tokio::task::JoinError), + #[error("certified Git snapshot is unavailable")] + Serving(#[from] crate::packs::publication::ServingReadError), + #[error("certified Git snapshot owner is unavailable")] + ServingOwner(#[from] crate::packs::publication::ServingOwnerError), } #[derive(Clone, Copy, PartialEq, Eq)] struct Node { @@ -119,6 +120,7 @@ pub(crate) struct FilePreview { pub(crate) struct Reader { repository: Arc, admission: Arc, + snapshot: Option, bytes: u64, entries: usize, } @@ -127,10 +129,35 @@ impl Reader { Self { repository, admission, + snapshot: None, bytes: 0, entries: 0, } } + async fn bind(&mut self, actor: ReadIdentity<'_>) -> Result<(), ReadError> { + if self.snapshot.is_some() { + return Err(ReadError::Malformed); + } + self.snapshot = Some(self.repository.serving_snapshot(actor).await?); + Ok(()) + } + fn snapshot(&self) -> Result<&crate::packs::publication::ServingSnapshot, ReadError> { + self.snapshot.as_ref().ok_or(ReadError::Malformed) + } + async fn header(&self, oid: Oid) -> Result { + if oid.format() != self.repository.object_format() { + return Err(ReadError::Invalid); + } + if oid.is_zero() { + return Err(ReadError::Missing); + } + self.snapshot()? + .headers(&[oid]) + .await? + .pop() + .flatten() + .ok_or(ReadError::Missing) + } async fn authorize<'a>( &self, actor: impl Into>, @@ -207,6 +234,7 @@ impl Reader { let actor = actor.into(); let cursor = after.map(path).transpose()?; let revision = self.authorize(actor, number, &target).await?; + self.bind(actor).await?; let (base, source) = (oid(&revision.base_oid)?, oid(&revision.source_oid)?); let (merge_base, before, after) = self.roots(base, source).await?; let changes = self.changes(before, after).await?; @@ -249,6 +277,7 @@ impl Reader { let actor = actor.into(); let path = path(encoded_path)?; let revision = self.authorize(actor, number, &target).await?; + self.bind(actor).await?; let (base, source) = (oid(&revision.base_oid)?, oid(&revision.source_oid)?); let (merge_base, before, after) = self.roots(base, source).await?; let root = match side { @@ -284,37 +313,11 @@ impl Reader { }) } async fn size(&self, oid: Oid, kind: ObjectKind) -> Result { - let result = self - .repository - .sql - .query( - None, - SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT kind, size FROM objects WHERE oid = ?1".into(), - parameters: vec![SqlValue::Blob(oid.to_vec())], - }], - }, - ) - .await?; - let Some([SqlValue::Text(stored_kind), SqlValue::Integer(size)]) = result - .output - .first() - .and_then(|set| set.rows.first()) - .map(Vec::as_slice) - else { - return Err(ReadError::Malformed); - }; - let expected = match kind { - ObjectKind::Blob => "blob", - ObjectKind::Tree => "tree", - ObjectKind::Commit => "commit", - ObjectKind::Tag => "tag", - }; - if stored_kind != expected { + let object = self.header(oid).await?.object; + if object.kind != kind { return Err(ReadError::Malformed); } - u64::try_from(*size).map_err(|_| ReadError::Malformed) + Ok(object.size) } async fn body(&mut self, oid: Oid, kind: ObjectKind) -> Result, ReadError> { let size = self.size(oid, kind).await?; @@ -322,13 +325,16 @@ impl Reader { return Err(ReadError::TooLarge); } self.bytes += size; - let (stored_kind, body) = self - .repository - .object(oid, None) - .await? - .output - .ok_or(ReadError::Malformed)?; - if stored_kind != kind || body.len() as u64 != size { + let body = self + .snapshot()? + .body(oid, MAX_OBJECT_BYTES as usize) + .await + .map_err(|error| match error { + crate::packs::publication::ServingReadError::TooLarge => ReadError::TooLarge, + error => ReadError::Serving(error), + })? + .ok_or(ReadError::Missing)?; + if body.len() as u64 != size { return Err(ReadError::Malformed); } Ok(body) diff --git a/crates/canopy-server/src/git_read/patch/mod.rs b/crates/canopy-server/src/git_read/patch/mod.rs index 82d8b37b..3f890cd2 100644 --- a/crates/canopy-server/src/git_read/patch/mod.rs +++ b/crates/canopy-server/src/git_read/patch/mod.rs @@ -106,6 +106,7 @@ impl Reader { let actor = actor.into(); let path = path(encoded_path)?; let revision = self.authorize(actor, number, &target).await?; + self.bind(actor).await?; let (merge_base, old_tree, new_tree) = self .roots(oid(&revision.base_oid)?, oid(&revision.source_oid)?) .await?; diff --git a/crates/canopy-server/src/lib.rs b/crates/canopy-server/src/lib.rs index cfa08361..5eadc5fc 100644 --- a/crates/canopy-server/src/lib.rs +++ b/crates/canopy-server/src/lib.rs @@ -84,22 +84,10 @@ pub(crate) fn replica_limits(database: u64, capture: u64) -> cellule_ltx::Limits } } -const SCHEMA: &str = include_str!("schema.sql"); -const COMMANDS: [OperationDescriptor; 9] = [ - operation(1), - operation_with_codec(3, 4), - operation_with_codec(4, 6), - OperationDescriptor { - input_limit: object_batch::INPUT_LIMIT, - ..operation_with_codec(5, 5) - }, - operation_with_codec(6, 3), - operation_with_codec(7, 2), - operation_with_codec(8, 2), - operation_with_codec(9, 4), - operation_with_codec(10, 2), -]; -const QUERIES: [OperationDescriptor; 1] = [operation(2)]; +/// Fresh packed schema shared by release migrations and actual Cell activation. +pub const REPOSITORY_SCHEMA: &str = packs::publication::SCHEMA; +const SCHEMA: &str = REPOSITORY_SCHEMA; +use packs::publication::registry::{COMMANDS, QUERIES}; const fn operation(id: u32) -> OperationDescriptor { operation_with_codec(id, 1) @@ -184,6 +172,12 @@ impl CellModule for RepositoryModule { source_digest: { let mut source = blake3::Hasher::new(); source.update(include_bytes!("lib.rs")); + source.update(include_bytes!("deployment/mod.rs")); + source.update(include_bytes!("deployment/root.rs")); + source.update(include_bytes!("server/mod.rs")); + source.update(include_bytes!("server/lifecycle.rs")); + source.update(include_bytes!("admission.rs")); + source.update(include_bytes!("server/workspace/mod.rs")); source.update(include_bytes!("../../canopy-git-format/src/lib.rs")); source.update(include_bytes!( "../../canopy-git-format/src/pack_index/mod.rs" @@ -200,23 +194,41 @@ impl CellModule for RepositoryModule { source.update(include_bytes!("object_reads/mod.rs")); source.update(include_bytes!("pack_store.rs")); source.update(include_bytes!("git_objects/mod.rs")); + source.update(include_bytes!("git_cache/artifacts.rs")); + source.update(include_bytes!("git_cache/serving_refs.rs")); + source.update(include_bytes!("git_cache/cleanup.rs")); + source.update(include_bytes!("git_cache/mod.rs")); + source.update(include_bytes!("git_cache/maintenance.rs")); + source.update(include_bytes!("packs/catalog/native.rs")); + source.update(include_bytes!("packs/catalog/reader.rs")); + source.update(include_bytes!("packs/catalog/graph_spool.rs")); + source.update(include_bytes!("packs/catalog/files.rs")); source.update(include_bytes!("native_resources.rs")); source.update(include_bytes!("native_git.rs")); source.update(include_bytes!("native_git/process.rs")); source.update(include_bytes!("native_git/process/fence.rs")); source.update(include_bytes!("git_gateway/mod.rs")); + source.update(include_bytes!("git_gateway/candidates/mod.rs")); + source.update(include_bytes!("git_gateway/fetch.rs")); + source.update(include_bytes!("git_gateway/discovery.rs")); + source.update(include_bytes!("git_gateway/ssh.rs")); source.update(include_bytes!("git_gateway/preflight.rs")); source.update(include_bytes!("git_gateway/preflight/retention.rs")); source.update(include_bytes!("git_gateway/branch_policy.rs")); source.update(include_bytes!("git_gateway/push.rs")); source.update(include_bytes!("git_input/mod.rs")); source.update(include_bytes!("git_http/capture.rs")); + source.update(include_bytes!("git_http/mod.rs")); + source.update(include_bytes!("packs/verification/mod.rs")); + source.update(include_bytes!("packs/verification/physical.rs")); + source.update(include_bytes!("packs/verification/spool.rs")); source.update(include_bytes!("packs/wire_request.rs")); source.update(include_bytes!("packs/input_artifact.rs")); source.update(include_bytes!("packs/directory/index/mod.rs")); source.update(include_bytes!("packs/directory/index/record.rs")); source.update(include_bytes!("packs/directory/index/codec.rs")); source.update(include_bytes!("packs/directory/index/cursor.rs")); + source.update(include_bytes!("packs/directory/index/changes.rs")); source.update(include_bytes!("packs/directory/index/update.rs")); source.update(include_bytes!("packs/directory/index/bulk.rs")); source.update(include_bytes!("packs/directory/index/rewrite.rs")); @@ -253,6 +265,12 @@ impl CellModule for RepositoryModule { source.update(include_bytes!("packs/publication/ref_proof.rs")); source.update(include_bytes!("packs/publication/ref_snapshot.rs")); source.update(include_bytes!("packs/publication/initialization.rs")); + source.update(include_bytes!( + "packs/publication/coordinator/initialization.rs" + )); + source.update(include_bytes!( + "packs/publication/recovery/initialization.rs" + )); source.update(include_bytes!("packs/publication/ref_policy/mod.rs")); source.update(include_bytes!("packs/publication/ref_policy/codec.rs")); source.update(include_bytes!("packs/publication/ref_policy/commands.rs")); @@ -263,7 +281,19 @@ impl CellModule for RepositoryModule { )); source.update(include_bytes!("packs/publication/staging_service.rs")); source.update(include_bytes!("packs/publication/staging_receipt.rs")); + source.update(include_bytes!("packs/publication/admission_receipt.rs")); + source.update(include_bytes!("packs/publication/custody/mod.rs")); + source.update(include_bytes!("packs/publication/custody/codec.rs")); + source.update(include_bytes!("packs/publication/custody/commands.rs")); + source.update(include_bytes!("packs/publication/custody/dispatch.rs")); + source.update(include_bytes!("packs/publication/preparation_receipt.rs")); source.update(include_bytes!("packs/publication/coordinator.rs")); + source.update(include_bytes!( + "packs/publication/coordinator/serving_drain.rs" + )); + source.update(include_bytes!("packs/publication/coordinator/budget.rs")); + source.update(include_bytes!("packs/publication/scan.rs")); + source.update(include_bytes!("packs/publication/custody/scan.rs")); source.update(include_bytes!("packs/publication/coordinator/policy.rs")); source.update(include_bytes!("packs/publication/coordinator/roots.rs")); source.update(include_bytes!("packs/publication/coordinator/work.rs")); @@ -283,6 +313,38 @@ impl CellModule for RepositoryModule { )); source.update(include_bytes!("packs/publication/exact.rs")); source.update(include_bytes!("packs/publication/mod.rs")); + source.update(include_bytes!("packs/publication/serving.rs")); + source.update(include_bytes!("packs/publication/serving/codec.rs")); + source.update(include_bytes!("packs/publication/serving/commands.rs")); + source.update(include_bytes!("packs/publication/serving/command_owner.rs")); + source.update(include_bytes!("packs/publication/serving/session.rs")); + source.update(include_bytes!("packs/publication/serving/session/reads.rs")); + source.update(include_bytes!("packs/publication/serving/session/body.rs")); + source.update(include_bytes!("packs/publication/serving/session/edges.rs")); + source.update(include_bytes!( + "packs/publication/serving/session/workspace.rs" + )); + source.update(include_bytes!( + "packs/publication/serving/session/native_base.rs" + )); + source.update(include_bytes!("git_read/mod.rs")); + source.update(include_bytes!("git_read/browse.rs")); + source.update(include_bytes!("git_read/graph.rs")); + source.update(include_bytes!("git_read/trees.rs")); + source.update(include_bytes!("git_read/patch/mod.rs")); + source.update(include_bytes!("packs/publication/serving/session/refs.rs")); + source.update(include_bytes!("packs/publication/serving/lifecycle.rs")); + source.update(include_bytes!("packs/publication/serving/pool.rs")); + source.update(include_bytes!( + "packs/publication/serving/session/handoff.rs" + )); + source.update(include_bytes!("packs/publication/serving/ownership.rs")); + source.update(include_bytes!("packs/publication/serving/schema.sql")); + source.update(include_bytes!("packs/publication/registry.rs")); + source.update(include_bytes!("server/catalog_initialization.rs")); + source.update(include_bytes!("server/residency/mod.rs")); + source.update(include_bytes!("server/residency/recovery.rs")); + source.update(include_bytes!("server/peer.rs")); source.update(include_bytes!("packs/publication/codec.rs")); source.update(include_bytes!("packs/publication/sql.rs")); source.update(include_bytes!("packs/publication/schema.sql")); @@ -348,14 +410,7 @@ impl CellModule for RepositoryModule { fn register(self, registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { register_sql::(registry)?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::() + packs::publication::register(registry) } } @@ -394,6 +449,7 @@ pub struct RepositoryCell { // Gateways own the cache lifetime. Sharing the reader through a weak // reference must not retain its original disk budget after gateway eviction. pack_readers: std::sync::Mutex>>, + serving: std::sync::Mutex>>, } impl RepositoryCell { @@ -421,9 +477,40 @@ impl RepositoryCell { application: application.clone(), target, pack_readers: std::sync::Mutex::new(Vec::new()), + serving: std::sync::Mutex::new(None), }) } + pub(crate) fn attach_serving(&self, pool: &std::sync::Arc) { + *self.serving.lock().expect("repository serving pool") = + Some(std::sync::Arc::downgrade(pool)); + } + /// Borrow the resident's certified joint generation; a detached caller + /// cannot abandon its acquisition or extend a released residency. + pub async fn serving_snapshot( + &self, + actor: ReadIdentity<'_>, + ) -> std::result::Result< + packs::publication::ServingSnapshot, + packs::publication::ServingOwnerError, + > { + actor + .validate() + .map_err(packs::publication::ServingReadError::Capability)?; + let pool = self + .serving + .lock() + .expect("repository serving pool") + .as_ref() + .and_then(std::sync::Weak::upgrade) + .ok_or(packs::publication::ServingReadError::Inactive)?; + let actor = match actor { + ReadIdentity::Anonymous => None, + ReadIdentity::Account(value) => Some(value.to_owned()), + }; + pool.snapshot(actor).await + } + /// Prepares bounded graph certificates, then publishes one all-or-none ref plan. pub async fn finalize_push( &self, diff --git a/crates/canopy-server/src/object_batch/mod.rs b/crates/canopy-server/src/object_batch/mod.rs index de6f4d22..f698ad30 100644 --- a/crates/canopy-server/src/object_batch/mod.rs +++ b/crates/canopy-server/src/object_batch/mod.rs @@ -18,6 +18,7 @@ pub(crate) const MAX_OBJECTS: usize = 128; // Publication amortizes durable commits independently of bounded read pages. pub(crate) const MAX_BATCH_OBJECTS: usize = 2048; pub(crate) const VERIFY_BATCH_BYTES: u64 = 64 * 1024 * 1024; +#[cfg(test)] pub(crate) const INPUT_LIMIT: u32 = 4 * 1024 * 1024; // Leave room for record metadata inside Cellule's bounded command wire format. const INLINE_BATCH_BYTES: usize = 3 * 1024 * 1024; diff --git a/crates/canopy-server/src/object_reads/mod.rs b/crates/canopy-server/src/object_reads/mod.rs index e8d6dee2..5562563d 100644 --- a/crates/canopy-server/src/object_reads/mod.rs +++ b/crates/canopy-server/src/object_reads/mod.rs @@ -1,8 +1,8 @@ //! Bounded immutable object pages for cold cache hydration. use cellule_runtime::{ - Error, InvocationError, Observed, Receipt, primitives::sql::SqlBatch, - primitives::sql::SqlResultSet, primitives::sql::SqlStatement, primitives::sql::SqlValue, + Error, InvocationError, Observed, primitives::sql::SqlBatch, primitives::sql::SqlResultSet, + primitives::sql::SqlStatement, primitives::sql::SqlValue, }; use crate::{ @@ -12,11 +12,8 @@ use crate::{ pub(crate) struct ObjectHeaders { pub(crate) objects: Observed>, - pub(crate) through: i64, } -const CHANGED_HEADERS: &str = "SELECT sequence, oid, CASE WHEN storage = 'inline' THEN size ELSE 0 END FROM objects WHERE sequence > ?1 AND sequence <= ?2 ORDER BY sequence LIMIT ?3"; - impl RepositoryCell { /// Reads at most 128 objects and 768 KiB of inline bodies, in OID order. /// @@ -50,55 +47,9 @@ impl RepositoryCell { Ok(self.object_records(headers.objects).await?.output) } - pub(crate) async fn object_high_water( - &self, - ) -> Result, InvocationError>> { - let result = self - .sql - .query( - None, - SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT COALESCE(MAX(sequence), 0) FROM objects".into(), - parameters: vec![], - }], - }, - ) - .await?; - let row = result.output.first().and_then(|set| set.rows.first()); - let Some([SqlValue::Integer(sequence)]) = row.map(Vec::as_slice) else { - return Err(InvocationError::NotStarted(Error::Command( - "invalid object high water", - ))); - }; - Ok(Observed { - output: *sequence, - receipt: result.receipt, - }) - } - - pub(crate) async fn object_headers( - &self, - after: i64, - high_water: &Observed, - ) -> Result>> { - self.read_object_headers( - Some(high_water.receipt), - SqlStatement { - sql: CHANGED_HEADERS.into(), - parameters: vec![ - SqlValue::Integer(after), - SqlValue::Integer(high_water.output), - SqlValue::Integer(MAX_OBJECTS as i64), - ], - }, - ) - .await - } - async fn read_object_headers( &self, - minimum: Option, + minimum: Option, statement: SqlStatement, ) -> Result>> { let headers = self @@ -114,13 +65,12 @@ impl RepositoryCell { .output .first() .ok_or_else(|| InvocationError::NotStarted(Error::Command("missing object headers")))?; - let (ids, through) = decode_headers(&rows.rows).map_err(InvocationError::NotStarted)?; + let (ids, _) = decode_headers(&rows.rows).map_err(InvocationError::NotStarted)?; Ok(ObjectHeaders { objects: Observed { output: ids, receipt: headers.receipt, }, - through, }) } @@ -290,6 +240,3 @@ fn decode_object(row: Vec) -> cellule_runtime::Result { }; Ok(StoredObject { oid, kind, storage }) } - -#[cfg(test)] -mod tests; diff --git a/crates/canopy-server/src/object_reads/tests.rs b/crates/canopy-server/src/object_reads/tests.rs deleted file mode 100644 index 384d9352..00000000 --- a/crates/canopy-server/src/object_reads/tests.rs +++ /dev/null @@ -1,127 +0,0 @@ -use super::*; -use cellule_ltx::rusqlite::{Connection, StatementStatus, params}; - -type Result = std::result::Result>; - -fn database() -> Result { - let db = Connection::open_in_memory()?; - db.execute_batch(crate::SCHEMA)?; - Ok(db) -} - -fn insert(db: &Connection, body: &[u8]) -> Result { - let oid = object_id(crate::ObjectFormat::Sha1, ObjectKind::Blob, body); - db.execute( - "INSERT INTO objects (oid, kind, size, digest, storage, body) VALUES (?1, 'blob', ?2, ?3, 'inline', ?4) ON CONFLICT(oid) DO NOTHING", - params![oid.as_ref(), body.len() as i64, blake3::hash(body).as_bytes().as_slice(), body], - )?; - Ok(oid) -} - -fn high_water(db: &Connection) -> Result { - Ok(db.query_row( - "SELECT COALESCE(MAX(sequence), 0) FROM objects", - [], - |row| row.get(0), - )?) -} - -fn page(db: &Connection, after: i64, high: i64) -> Result<(Vec, i64)> { - let mut statement = db.prepare(CHANGED_HEADERS)?; - let rows = statement - .query_map(params![after, high, MAX_OBJECTS as i64], |row| { - Ok(vec![ - SqlValue::Integer(row.get(0)?), - SqlValue::Blob(row.get(1)?), - SqlValue::Integer(row.get(2)?), - ]) - })? - .collect::, _>>()?; - // An indexed range must neither walk the old history nor sort it anew. - assert_eq!(statement.get_status(StatementStatus::FullscanStep), 0); - assert_eq!(statement.get_status(StatementStatus::Sort), 0); - Ok(decode_headers(&rows)?) -} - -#[test] -fn insertion_cursor_finds_lower_oids_and_excludes_later_publications() -> Result { - let db = database()?; - let mut bodies: Vec<_> = (0..260) - .map(|n| format!("cursor-{n}").into_bytes()) - .collect(); - bodies.sort_by_key(|body| { - std::cmp::Reverse(object_id(crate::ObjectFormat::Sha1, ObjectKind::Blob, body)) - }); - let mut expected = Vec::new(); - for body in &bodies[..259] { - expected.push(insert(&db, body)?); - } - let high = high_water(&db)?; - let late = insert(&db, &bodies[259])?; - assert!(late < *expected.last().ok_or("missing object")?); - let mut after = 0; - let mut actual = Vec::new(); - while after < high { - let (ids, through) = page(&db, after, high)?; - assert!(!ids.is_empty()); - assert!(ids.len() <= MAX_OBJECTS); - actual.extend(ids); - after = through; - } - assert_eq!(actual, expected); - assert!(page(&db, after, high)?.0.is_empty()); - assert_eq!(page(&db, after, high_water(&db)?)?.0, vec![late]); - Ok(()) -} - -#[test] -fn byte_limited_page_advances_only_over_the_selected_prefix() -> Result { - let db = database()?; - let first = insert(&db, &vec![1; INLINE_OBJECT_LIMIT / 2])?; - let second = insert(&db, &vec![2; INLINE_OBJECT_LIMIT])?; - let third = insert(&db, b"third")?; - let high = high_water(&db)?; - let (ids, after) = page(&db, 0, high)?; - assert_eq!(ids, vec![first]); - let (ids, after) = page(&db, after, high)?; - assert_eq!(ids, vec![second]); - assert_eq!(page(&db, after, high)?.0, vec![third]); - Ok(()) -} - -#[test] -fn duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts() -> Result { - let db = database()?; - let original = insert(&db, b"original")?; - let after = high_water(&db)?; - insert(&db, b"original")?; - assert_eq!(high_water(&db)?, after); - db.execute_batch("SAVEPOINT failed_batch")?; - insert(&db, b"rolled back")?; - db.execute_batch("ROLLBACK TO failed_batch; RELEASE failed_batch")?; - assert_eq!(high_water(&db)?, after); - db.execute("DELETE FROM objects WHERE oid = ?1", [original.as_ref()])?; - // The product has no collector yet. Never reusing a committed cursor also - // protects a future fenced collection from hiding newly inserted rows. - let next = insert(&db, b"after deletion")?; - assert!(high_water(&db)? > after); - assert_eq!(page(&db, after, high_water(&db)?)?.0, vec![next]); - Ok(()) -} - -#[test] -fn small_increment_uses_bounded_sql_work_after_large_history() -> Result { - let db = database()?; - db.execute_batch("BEGIN")?; - for n in 0..10_000 { - insert(&db, format!("history-{n}").as_bytes())?; - } - db.execute_batch("COMMIT")?; - let after = high_water(&db)?; - let mut expected = Vec::new(); - for body in [b"one".as_slice(), b"two", b"three"] { - expected.push(insert(&db, body)?); - } - assert_eq!(page(&db, after, high_water(&db)?)?.0, expected); - Ok(()) -} diff --git a/crates/canopy-server/src/pack_store.rs b/crates/canopy-server/src/pack_store.rs index 6968a69b..368e303b 100644 --- a/crates/canopy-server/src/pack_store.rs +++ b/crates/canopy-server/src/pack_store.rs @@ -21,7 +21,6 @@ pub(crate) struct PackRecord { pub pack: LargeBlobReference, pub index: LargeBlobReference, pub approved: bool, - pub covered_through: i64, } pub(crate) struct PackReader { @@ -72,12 +71,6 @@ impl PackReader { cache.as_ref().ok_or(GatewayError::MalformedCache)?, )) } - pub(crate) async fn replace(&self, old: &Arc, next: Arc) { - let mut cache = self.cache.lock().await; - if cache.as_ref().is_some_and(|cache| Arc::ptr_eq(cache, old)) { - *cache = Some(next); - } - } pub(crate) async fn install( &self, cache: &Arc, @@ -155,6 +148,7 @@ impl PackReader { process, output, oid, + #[cfg(test)] size, remaining: size, expected: digest, @@ -219,6 +213,7 @@ pub(crate) struct NativePackedRead { process: GitProcess>, output: tokio::process::ChildStdout, pub(crate) oid: ObjectId, + #[cfg(test)] pub(crate) size: u64, remaining: u64, expected: [u8; 32], @@ -294,7 +289,7 @@ fn decode(row: &[SqlValue]) -> Result { SqlValue::Blob(index_digest), SqlValue::Blob(index_sha), SqlValue::Integer(approved), - SqlValue::Integer(covered_through), + SqlValue::Integer(_covered_through), ] = row else { return Err(invalid()); @@ -314,7 +309,6 @@ fn decode(row: &[SqlValue]) -> Result { sha256: index_sha.as_slice().try_into().map_err(|_| invalid())?, }, approved: *approved == 1, - covered_through: *covered_through, }) } impl RepositoryCell { @@ -339,17 +333,6 @@ impl RepositoryCell { .ok_or_else(invalid)?, ) } - pub(crate) async fn approved_packs(&self, after: &[u8]) -> Result, ReadError> { - let result = self.sql.query(None, SqlBatch { statements: vec![SqlStatement { sql: format!("SELECT {COLUMNS} FROM git_packs WHERE approved = 1 AND sha256 > ?1 ORDER BY sha256 LIMIT 128"), parameters: vec![SqlValue::Blob(after.to_vec())] }] }).await?; - result - .output - .first() - .ok_or_else(invalid)? - .rows - .iter() - .map(|row| decode(row)) - .collect() - } pub(crate) async fn register_pack( &self, identity: MutationIdentity, diff --git a/crates/canopy-server/src/packs/catalog/files.rs b/crates/canopy-server/src/packs/catalog/files.rs index 5be6fc8e..76606ce9 100644 --- a/crates/canopy-server/src/packs/catalog/files.rs +++ b/crates/canopy-server/src/packs/catalog/files.rs @@ -46,6 +46,7 @@ pub struct CatalogFileStats { /// Shared across catalog generations in one worker. Creating a loader does not /// certify its catalogs, grant access, or pin remote generations against GC. pub struct CatalogFiles { + native: Option, root: Arc, store: Arc, format: ObjectFormat, @@ -121,6 +122,7 @@ impl CatalogFiles { .prefix("canopy-catalog-files-") .tempdir_in(workspace)?; Ok(Self { + native: None, root: Arc::new(root), store, format, @@ -133,6 +135,91 @@ impl CatalogFiles { downloaded_files: AtomicU64::new(0), }) } + /// Configure native serving with the node's shared resource scope. Metadata + /// preparation alone does not need or implicitly create a native service. + pub(crate) fn with_native(mut self, native: crate::native_resources::NativeScope) -> Self { + self.native = Some(super::native::NativeFiles::new( + Arc::clone(&self.root), + self.budget.clone(), + Arc::clone(&self.store), + self.format, + native, + )); + self + } + pub(in crate::packs) async fn workspace( + &self, + owner: crate::git_objects::ReadOwner, + cleanup: crate::git_objects::ReadOwner, + head: String, + ) -> Result, super::native::NativeReadError> { + self.native + .as_ref() + .ok_or(super::native::NativeReadError::Unavailable)? + .workspace(owner, cleanup, head) + .await + } + pub(in crate::packs) async fn write_workspace( + &self, + owner: crate::git_objects::ReadOwner, + cleanup: crate::git_objects::ReadOwner, + head: String, + objects: Arc, + ) -> Result, super::native::NativeReadError> { + self.native + .as_ref() + .ok_or(super::native::NativeReadError::Unavailable)? + .workspace_with_objects(owner, cleanup, head, Some(objects)) + .await + } + pub(in crate::packs) async fn install_workspace( + &self, + cache: Arc, + source: super::super::sources::NativePackDescriptor, + owner: crate::git_objects::ReadOwner, + ) -> Result<(), super::native::NativeReadError> { + self.native + .as_ref() + .ok_or(super::native::NativeReadError::Unavailable)? + .install(cache, source, owner) + .await + } + pub(in crate::packs) async fn graph_spool( + &self, + maximum: u64, + cache_kib: u32, + owner: crate::git_objects::ReadOwner, + cleanup: crate::git_objects::ReadOwner, + ) -> Result>, MetadataError> { + let (root, budget) = (self.root.clone(), self.budget.clone()); + tokio::task::spawn_blocking(move || { + let _owner = owner; + Ok(Arc::new(Mutex::new(super::graph_spool::GraphSpool::new( + root, budget, maximum, cache_kib, cleanup, + )?))) + }) + .await? + } + pub(in crate::packs) async fn body( + &self, + object: ResolvedObject, + limit: usize, + owner: crate::git_objects::ReadOwner, + ) -> Result, super::native::NativeReadError> { + self.native + .as_ref() + .ok_or(super::native::NativeReadError::Unavailable)? + .body(object, limit, owner) + .await + } + pub fn native_stats( + &self, + ) -> Result, super::native::NativeReadError> { + self.native + .as_ref() + .map(super::native::NativeFiles::stats) + .transpose() + } pub fn stats(&self) -> Result { Ok(CatalogFileStats { open_files: self.limits.open_files - self.slots.available_permits() as u32, diff --git a/crates/canopy-server/src/packs/catalog/graph_spool.rs b/crates/canopy-server/src/packs/catalog/graph_spool.rs new file mode 100644 index 00000000..da9cad15 --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/graph_spool.rs @@ -0,0 +1,198 @@ +//! Disposable, admitted graph frontier and source deduplication. No repository +//! SQL rows, history-sized Rust sets, or authority are stored here. +use crate::git_objects::ReadOwner; +use crate::packs::{ + metadata::{AdmittedFile, MetadataError, growth}, + sources::NativePackDescriptor, +}; +use crate::{ObjectId, ObjectKind}; +use cellule_ltx::DiskBudget; +use rusqlite::{Connection, OptionalExtension, params}; +use std::sync::Arc; + +type PackBinding = (Vec, Vec, u64, u64, u32); + +pub(in crate::packs) struct GraphSpool { + db: Connection, + file: AdmittedFile, + maximum: u64, + // Drops after SQLite, journal and file cleanup, including blocking jobs. + _owner: ReadOwner, +} +impl GraphSpool { + pub(in crate::packs) fn new( + root: Arc, + budget: DiskBudget, + maximum: u64, + cache_kib: u32, + owner: ReadOwner, + ) -> Result { + let reservation = growth::reserve(&budget, maximum)?; + let mut file = AdmittedFile::new( + tempfile::Builder::new() + .prefix("canopy-graph-") + .tempfile_in(root.path())?, + reservation, + ); + file.retain_workspace(root); + let mut db = Connection::open(file.file().path())?; + db.execute_batch("PRAGMA page_size=4096; PRAGMA journal_mode=DELETE; PRAGMA synchronous=FULL; PRAGMA foreign_keys=ON; PRAGMA trusted_schema=OFF; PRAGMA mmap_size=0; PRAGMA temp_store=MEMORY;")?; + db.pragma_update(None, "cache_size", -(i64::from(cache_kib)))?; + growth::configure(&db, &mut file)?; + growth::transaction(&mut db, &mut file, maximum, |tx| { + tx.execute_batch("CREATE TABLE nodes(oid BLOB PRIMARY KEY,kind TEXT,done INTEGER NOT NULL DEFAULT 0 CHECK(done IN (0,1))) WITHOUT ROWID; CREATE INDEX pending ON nodes(done,oid); CREATE TABLE packs(checksum BLOB PRIMARY KEY,pack BLOB NOT NULL,idx BLOB NOT NULL,pack_bytes INTEGER NOT NULL,idx_bytes INTEGER NOT NULL,objects INTEGER NOT NULL) WITHOUT ROWID;")?; + Ok::<_, MetadataError>(()) + })?; + Ok(Self { + db, + file, + maximum, + _owner: owner, + }) + } + pub(in crate::packs) fn add( + &mut self, + ids: &[(ObjectId, Option)], + ) -> Result<(), MetadataError> { + if ids.len() > 512 { + return Err(MetadataError::Limit); + } + growth::transaction(&mut self.db, &mut self.file, self.maximum, |tx| { + let mut insert = tx.prepare_cached( + "INSERT INTO nodes(oid,kind) VALUES(?1,?2) ON CONFLICT DO NOTHING", + )?; + let mut read = tx.prepare_cached("SELECT kind FROM nodes WHERE oid=?1")?; + let mut update = + tx.prepare_cached("UPDATE nodes SET kind=?2 WHERE oid=?1 AND kind IS NULL")?; + for (id, kind) in ids { + insert.execute(params![id.as_ref(), kind.map(ObjectKind::git_name)])?; + if let Some(kind) = kind { + let existing: Option = read.query_row([id.as_ref()], |r| r.get(0))?; + if existing + .as_deref() + .is_some_and(|old| old != kind.git_name()) + { + return Err(MetadataError::Integrity); + } + update.execute(params![id.as_ref(), kind.git_name()])?; + } + } + Ok::<_, MetadataError>(()) + }) + } + pub(in crate::packs) fn pending( + &self, + ) -> Result)>, MetadataError> { + let mut query = self + .db + .prepare_cached("SELECT oid,kind FROM nodes WHERE done=0 ORDER BY oid LIMIT 128")?; + query + .query_map([], |row| { + let id: Vec = row.get(0)?; + let name: Option = row.get(1)?; + let kind = name + .as_deref() + .map(|name| match name { + "blob" => Ok(ObjectKind::Blob), + "tree" => Ok(ObjectKind::Tree), + "commit" => Ok(ObjectKind::Commit), + "tag" => Ok(ObjectKind::Tag), + _ => Err(rusqlite::Error::InvalidQuery), + }) + .transpose()?; + Ok(( + ObjectId::try_from(id).map_err(|_| rusqlite::Error::InvalidQuery)?, + kind, + )) + })? + .collect::>() + .map_err(Into::into) + } + pub(in crate::packs) fn done( + &mut self, + ids: &[(ObjectId, Option)], + ) -> Result<(), MetadataError> { + growth::transaction(&mut self.db, &mut self.file, self.maximum, |tx| { + let mut update = + tx.prepare_cached("UPDATE nodes SET done=1 WHERE oid=?1 AND kind IS NOT NULL")?; + for (id, _) in ids { + if update.execute([id.as_ref()])? != 1 { + return Err(MetadataError::Integrity); + } + } + Ok::<_, MetadataError>(()) + }) + } + pub(in crate::packs) fn pack_seen( + &self, + p: NativePackDescriptor, + ) -> Result { + let old: Option = self + .db + .query_row( + "SELECT pack,idx,pack_bytes,idx_bytes,objects FROM packs WHERE checksum=?1", + [p.git_checksum.as_ref()], + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?, r.get(4)?)), + ) + .optional()?; + let Some((pack, idx, pack_bytes, idx_bytes, objects)) = old else { + return Ok(false); + }; + if pack != p.pack.digest + || idx != p.index.digest + || pack_bytes != p.pack.size + || idx_bytes != p.index.size + || objects != p.object_count + { + return Err(MetadataError::IdentityConflict); + } + Ok(true) + } + pub(in crate::packs) fn imported( + &mut self, + p: NativePackDescriptor, + ) -> Result<(), MetadataError> { + growth::transaction(&mut self.db, &mut self.file, self.maximum, |tx| { + tx.execute( + "INSERT INTO packs VALUES(?1,?2,?3,?4,?5,?6)", + params![ + p.git_checksum.as_ref(), + p.pack.digest.as_slice(), + p.index.digest.as_slice(), + p.pack.size, + p.index.size, + p.object_count + ], + )?; + Ok::<_, MetadataError>(()) + }) + } + pub(in crate::packs) fn contains(&self, ids: &[ObjectId]) -> Result, MetadataError> { + let mut q = self + .db + .prepare_cached("SELECT done FROM nodes WHERE oid=?1")?; + ids.iter() + .map(|id| { + Ok(q.query_row([id.as_ref()], |r| r.get::<_, bool>(0)) + .optional()? + .unwrap_or(false)) + }) + .collect() + } + #[cfg(test)] + fn counts(&self) -> Result<(u64, u64), MetadataError> { + Ok(( + self.db + .query_row("SELECT count(*) FROM nodes WHERE done=1", [], |r| r.get(0))?, + self.db + .query_row("SELECT count(*) FROM packs", [], |r| r.get(0))?, + )) + } + #[cfg(test)] + fn path(&self) -> &std::path::Path { + self.file.file().path() + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/catalog/graph_spool/tests.rs b/crates/canopy-server/src/packs/catalog/graph_spool/tests.rs new file mode 100644 index 00000000..bbe5e6e4 --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/graph_spool/tests.rs @@ -0,0 +1,191 @@ +use super::*; +use crate::ObjectFormat; +use canopy_object_storage::artifact::ArtifactDescriptor; +type Result = std::result::Result>; + +fn id(format: ObjectFormat, n: u64) -> ObjectId { + let mut bytes = vec![0; format.bytes()]; + bytes[format.bytes() - 8..].copy_from_slice(&n.to_be_bytes()); + ObjectId::try_from(bytes).unwrap() +} +fn spool(budget: &DiskBudget, maximum: u64) -> Result { + Ok(GraphSpool::new( + Arc::new(tempfile::TempDir::new()?), + budget.clone(), + maximum, + 16, + Arc::new(()), + )?) +} + +#[test] +fn frontier_is_indexed_paged_deduplicated_and_only_complete_members_are_visible() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let budget = DiskBudget::new(3 << 20); + let mut s = spool(&budget, 1 << 20)?; + let path = s.path().to_owned(); + let ids: Vec<_> = (1..=2500) + .map(|n| (id(format, n), Some(ObjectKind::Blob))) + .collect(); + for page in ids.chunks(512).rev() { + s.add(page)?; + s.add(page)?; + } + assert!(budget.used() > growth::INITIAL_BYTES * 3); + assert!(budget.used() < budget.capacity()); + let plan: String = s.db.query_row( + "EXPLAIN QUERY PLAN SELECT oid,kind FROM nodes WHERE done=0 ORDER BY oid LIMIT 128", + [], + |r| r.get(3), + )?; + assert!(plan.contains("pending"), "{plan}"); + assert!(!plan.contains("TEMP"), "{plan}"); + let mut completed = 0; + loop { + let page = s.pending()?; + if page.is_empty() { + break; + } + assert!(page.len() <= 128); + assert!(page.windows(2).all(|p| p[0].0 < p[1].0)); + assert_eq!(page[0].0, id(format, completed + 1)); + let keys: Vec<_> = page.iter().map(|(id, _)| *id).collect(); + assert!(s.contains(&keys)?.iter().all(|present| !present)); + s.done(&page)?; + assert!(s.contains(&keys)?.iter().all(|present| *present)); + completed += page.len() as u64; + } + assert_eq!(completed, 2500); + assert_eq!(s.counts()?, (2500, 0)); + assert_eq!( + s.contains(&[id(format, 2), id(format, 2501), id(format, 1)])?, + [true, false, true] + ); + { + use rusqlite::StatementStatus; + let mut lookup = s.db.prepare_cached("SELECT done FROM nodes WHERE oid=?1")?; + lookup.reset_status(StatementStatus::VmStep); + assert!(lookup.query_row([id(format, 1).as_ref()], |r| r.get::<_, bool>(0))?); + assert!(lookup.get_status(StatementStatus::VmStep) < 50); + assert_eq!(lookup.get_status(StatementStatus::FullscanStep), 0); + } + drop(s); + assert!(!path.exists()); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[test] +fn kind_conflict_and_incomplete_done_roll_back_the_entire_batch() -> Result { + let budget = DiskBudget::new(3 << 20); + let mut s = spool(&budget, 1 << 20)?; + let (a, b, c) = ( + id(ObjectFormat::Sha256, 1), + id(ObjectFormat::Sha256, 2), + id(ObjectFormat::Sha256, 3), + ); + s.add(&[(a, None), (b, Some(ObjectKind::Tree))])?; + assert!(matches!( + s.add(&[(c, Some(ObjectKind::Blob)), (b, Some(ObjectKind::Commit))]), + Err(MetadataError::Integrity) + )); + assert_eq!(s.pending()?, [(a, None), (b, Some(ObjectKind::Tree))]); + assert!(matches!( + s.done(&[(b, Some(ObjectKind::Tree)), (a, None)]), + Err(MetadataError::Integrity) + )); + assert_eq!(s.contains(&[a, b, c])?, [false, false, false]); + s.add(&[(a, Some(ObjectKind::Commit))])?; + s.done(&[(a, None), (b, None)])?; + assert_eq!(s.counts()?, (2, 0)); + Ok(()) +} + +#[test] +fn failed_growth_retains_prior_membership_without_a_partial_frontier() -> Result { + let budget = DiskBudget::new(growth::INITIAL_BYTES * 3); + let mut s = spool(&budget, 1 << 20)?; + let initial = id(ObjectFormat::Sha256, 1); + s.add(&[(initial, Some(ObjectKind::Commit))])?; + s.done(&[(initial, None)])?; + let mut accepted = 0; + loop { + let page: Vec<_> = (0..512) + .map(|n| { + ( + id(ObjectFormat::Sha256, accepted + n + 2), + Some(ObjectKind::Blob), + ) + }) + .collect(); + match s.add(&page) { + Ok(()) => accepted += 512, + Err(MetadataError::Budget(_)) => { + let count: u64 = + s.db.query_row("SELECT count(*) FROM nodes", [], |r| r.get(0))?; + assert_eq!(count, accepted + 1); + assert!(s.db.is_autocommit()); + assert_eq!( + s.contains(&[initial, page[0].0, page[511].0])?, + [true, false, false] + ); + break; + } + Err(e) => return Err(e.into()), + } + assert!(accepted < 10_000); + } + assert_eq!(budget.used(), growth::INITIAL_BYTES * 3); + drop(s); + assert_eq!(budget.used(), 0); + let denied = DiskBudget::new(growth::INITIAL_BYTES * 3 - 1); + assert!(spool(&denied, 1 << 20).is_err()); + assert_eq!(denied.used(), 0); + Ok(()) +} + +#[test] +fn physical_input_dedup_ignores_namespace_but_rejects_conflicting_bindings() -> Result { + let budget = DiskBudget::new(3 << 20); + let mut s = spool(&budget, 1 << 20)?; + let p = NativePackDescriptor { + repository: [1; 16], + operation: [2; 16], + format: ObjectFormat::Sha1, + git_checksum: id(ObjectFormat::Sha1, 1), + object_count: 1, + pack: ArtifactDescriptor { + size: 100, + digest: [3; 32], + manifest_digest: [4; 32], + }, + index: ArtifactDescriptor { + size: 1100, + digest: [5; 32], + manifest_digest: [6; 32], + }, + }; + assert!(!s.pack_seen(p)?); + s.imported(p)?; + let mut other = p; + other.operation = [7; 16]; + other.pack.manifest_digest = [8; 32]; + assert!(s.pack_seen(other)?); + for mutation in 0..5 { + let mut conflict = other; + match mutation { + 0 => conflict.pack.digest[0] ^= 1, + 1 => conflict.index.digest[0] ^= 1, + 2 => conflict.pack.size += 1, + 3 => conflict.index.size += 1, + _ => conflict.object_count += 1, + } + assert!(matches!( + s.pack_seen(conflict), + Err(MetadataError::IdentityConflict) + )); + } + assert_eq!(s.counts()?, (0, 1)); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/catalog/mod.rs b/crates/canopy-server/src/packs/catalog/mod.rs index 0ee499b6..5d4904aa 100644 --- a/crates/canopy-server/src/packs/catalog/mod.rs +++ b/crates/canopy-server/src/packs/catalog/mod.rs @@ -16,6 +16,9 @@ use canopy_object_storage::artifact::{ use std::sync::Arc; mod codec; +pub(in crate::packs) mod graph_spool; +mod native; +pub use native::{NativeFileStats, NativeReadError}; mod files; pub use files::{CatalogFileLimits, CatalogFileStats, CatalogFiles, MAX_OPEN_CATALOG_FILES}; mod reader; @@ -116,3 +119,6 @@ impl CatalogSnapshot { #[cfg(test)] pub(in crate::packs) mod tests; + +#[cfg(test)] +pub(crate) mod serving_fixture; diff --git a/crates/canopy-server/src/packs/catalog/native.rs b/crates/canopy-server/src/packs/catalog/native.rs new file mode 100644 index 00000000..72ed5052 --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/native.rs @@ -0,0 +1,306 @@ +//! Bounded shared immutable pack copies. Callers supply certified selection and +//! a physical generation guard; these files do not themselves grant authority. +use super::*; +use crate::{ + git_cache::{CacheError, CacheOwnership, GitCache}, + git_objects::{GitObjects, ObjectReadError, ReadOwner}, + native_resources::NativeScope, + packs::{metadata::MetadataError, sources::NativePackDescriptor}, +}; +use cellule_ltx::DiskBudget; +use std::{collections::VecDeque, sync::Mutex}; +use tokio::sync::{Mutex as AsyncMutex, OwnedSemaphorePermit, Semaphore}; + +const OPEN_PACKS: usize = 8; +const CACHED_PACKS: usize = 4; +const LOAD_STRIPES: usize = 16; +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NativeFileStats { + pub open_files: usize, + pub cached_files: usize, + pub cache_hits: u64, + pub downloaded_files: u64, +} + +#[derive(Debug, thiserror::Error)] +pub enum NativeReadError { + #[error("native read service is unavailable")] + Unavailable, + #[error("native read file admission exhausted")] + Capacity, + #[error("native cache creation failed")] + Cache(#[from] CacheError), + #[error("native pack transfer failed")] + Metadata(#[from] MetadataError), + #[error("native pack binding failed")] + Binding(#[from] IndexError), + #[error("native object verification failed")] + Object(#[from] ObjectReadError), + #[error("native read job failed")] + Task(#[from] tokio::task::JoinError), +} + +pub(super) struct NativeFiles { + root: Arc, + budget: DiskBudget, + store: Arc, + format: ObjectFormat, + native: NativeScope, + slots: Arc, + cache: Mutex)>>, + loads: [AsyncMutex<()>; LOAD_STRIPES], + hits: std::sync::atomic::AtomicU64, + downloads: std::sync::atomic::AtomicU64, +} +struct FileAdmission { + _root: Arc, + _slot: OwnedSemaphorePermit, +} +struct PackFile { + cache: Arc, + _admission: Arc, +} +impl NativeFiles { + pub(super) fn new( + root: Arc, + budget: DiskBudget, + store: Arc, + format: ObjectFormat, + native: NativeScope, + ) -> Self { + Self { + root, + budget, + store, + format, + native, + slots: Arc::new(Semaphore::new(OPEN_PACKS)), + cache: Mutex::new(VecDeque::new()), + loads: std::array::from_fn(|_| AsyncMutex::new(())), + hits: std::sync::atomic::AtomicU64::new(0), + downloads: std::sync::atomic::AtomicU64::new(0), + } + } + pub(super) fn stats(&self) -> Result { + Ok(NativeFileStats { + open_files: OPEN_PACKS - self.slots.available_permits(), + cached_files: self + .cache + .lock() + .map_err(|_| NativeReadError::Capacity)? + .len(), + cache_hits: self.hits.load(std::sync::atomic::Ordering::Relaxed), + downloaded_files: self.downloads.load(std::sync::atomic::Ordering::Relaxed), + }) + } + fn cached( + &self, + descriptor: NativePackDescriptor, + ) -> Result>, NativeReadError> { + let mut cache = self.cache.lock().map_err(|_| NativeReadError::Capacity)?; + let Some(at) = cache.iter().position(|(key, _)| *key == descriptor) else { + return Ok(None); + }; + let entry = cache.remove(at).ok_or(NativeReadError::Capacity)?; + let file = Arc::clone(&entry.1); + cache.push_back(entry); + self.hits.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + Ok(Some(file)) + } + async fn admit( + &self, + owner: ReadOwner, + bytes: u64, + ) -> Result, NativeReadError> { + loop { + if self.budget.available() >= bytes + && let Ok(slot) = Arc::clone(&self.slots).try_acquire_owned() + { + return Ok(Arc::new(FileAdmission { + _root: Arc::clone(&self.root), + _slot: slot, + })); + } + let evicted = { + let mut cache = self.cache.lock().map_err(|_| NativeReadError::Capacity)?; + cache + .iter() + .position(|(_, file)| Arc::strong_count(file) == 1) + .and_then(|at| cache.remove(at)) + }; + let Some(evicted) = evicted else { + return Err(NativeReadError::Capacity); + }; + let owner = Arc::clone(&owner); + tokio::task::spawn_blocking(move || { + let _owner = owner; + drop(evicted); + }) + .await?; + } + } + async fn load( + &self, + descriptor: NativePackDescriptor, + owner: ReadOwner, + ) -> Result, NativeReadError> { + descriptor.validate(self.store.repository(), self.format)?; + if let Some(file) = self.cached(descriptor)? { + return Ok(file); + } + let stripe = usize::from(descriptor.pack.digest[0]) % LOAD_STRIPES; + let _loading = self.loads[stripe].lock().await; + if let Some(file) = self.cached(descriptor)? { + return Ok(file); + } + let bytes = descriptor + .pack + .size + .checked_add(descriptor.index.size) + .and_then(|size| size.checked_add(4096)) + .ok_or(NativeReadError::Capacity)?; + if bytes > self.budget.capacity() { + return Err(NativeReadError::Capacity); + } + let admission = self.admit(Arc::clone(&owner), bytes).await?; + let lifetime: ReadOwner = Arc::new((Arc::clone(&owner), Arc::clone(&admission))); + let cache = GitCache::create_owned( + self.root.path().to_owned(), + self.budget.clone(), + "refs/heads/main", + self.format, + None, + self.native.clone(), + CacheOwnership { + work: Arc::clone(&lifetime), + cleanup: Some(admission.clone()), + }, + ) + .await?; + cache + .download_native_owned(&self.store, descriptor, Arc::clone(&lifetime)) + .await?; + let file = Arc::new(PackFile { + cache, + _admission: admission, + }); + let verify = Arc::clone(&file); + let claim = self + .native + .try_admit(crate::native_resources::NativeWork::Read) + .map_err(ObjectReadError::from)?; + tokio::task::spawn_blocking(move || { + let (_owner, _claim) = (lifetime, claim); + let pack = verify.cache.git_dir().join(format!( + "objects/pack/pack-{}.pack", + hex::encode(descriptor.git_checksum) + )); + descriptor.verify_files(&pack, &pack.with_extension("idx"))?; + Ok::<_, IndexError>(()) + }) + .await??; + self.downloads + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let evicted = { + let mut cache = self.cache.lock().map_err(|_| NativeReadError::Capacity)?; + cache.push_back((descriptor, Arc::clone(&file))); + if cache.len() > CACHED_PACKS { + cache.pop_front() + } else { + None + } + }; + if let Some(evicted) = evicted { + tokio::task::spawn_blocking(move || { + let _owner = owner; + drop(evicted); + }) + .await?; + } + Ok(file) + } + pub(super) async fn workspace( + &self, + owner: ReadOwner, + cleanup: ReadOwner, + head: String, + ) -> Result, NativeReadError> { + self.workspace_with_objects(owner, cleanup, head, None) + .await + } + pub(super) async fn workspace_with_objects( + &self, + owner: ReadOwner, + cleanup: ReadOwner, + head: String, + objects: Option>, + ) -> Result, NativeReadError> { + let admission = self.admit(owner.clone(), 4096).await?; + let lifetime: ReadOwner = Arc::new((cleanup, admission)); + Ok(GitCache::create_owned( + self.root.path().to_owned(), + self.budget.clone(), + &head, + self.format, + objects, + self.native.clone(), + CacheOwnership { + work: Arc::new((owner, lifetime.clone())), + cleanup: Some(lifetime), + }, + ) + .await?) + } + pub(super) async fn install( + &self, + cache: Arc, + descriptor: NativePackDescriptor, + owner: ReadOwner, + ) -> Result<(), NativeReadError> { + cache + .download_native_owned(&self.store, descriptor, owner.clone()) + .await?; + let claim = self + .native + .try_admit(crate::native_resources::NativeWork::Read) + .map_err(ObjectReadError::from)?; + tokio::task::spawn_blocking(move || { + let (_owner, _claim) = (owner, claim); + let pack = cache.git_dir().join(format!( + "objects/pack/pack-{}.pack", + hex::encode(descriptor.git_checksum) + )); + descriptor.verify_files(&pack, &pack.with_extension("idx"))?; + Ok::<_, IndexError>(()) + }) + .await??; + self.downloads + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + Ok(()) + } + pub(super) async fn body( + &self, + object: ResolvedObject, + limit: usize, + owner: ReadOwner, + ) -> Result, NativeReadError> { + // Metadata was selected and verified through the certified catalog. No + // native lookup is attempted for guessed or unpublished object IDs. + let expected = object.entry.header.object; + if expected.size > limit as u64 { + return Err(ObjectReadError::TooLarge.into()); + } + let file = self + .load(object.source.record.native(), Arc::clone(&owner)) + .await?; + let process_owner: ReadOwner = Arc::new((owner, Arc::clone(&file))); + let mut objects = + GitObjects::batch_owned(&file.cache.git_dir(), &self.native, process_owner)?; + let body = objects.read_verified(expected, limit).await?; + objects.finish().await?; + Ok(body) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/catalog/native/tests.rs b/crates/canopy-server/src/packs/catalog/native/tests.rs new file mode 100644 index 00000000..42a654fc --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/native/tests.rs @@ -0,0 +1,116 @@ +use super::*; +use crate::packs::verification::physical::tests::{Prepared, prepared_for_store}; +use object_store::memory::InMemory; +type Result = std::result::Result>; + +async fn packs(format: ObjectFormat, counts: &[usize]) -> Result> { + let provider = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), [1; 16])); + let mut inputs = Vec::new(); + for (n, count) in counts.iter().enumerate() { + let mut operation = *b"CANOPY0100000000"; + operation[8..].copy_from_slice(&(n as u64 + 700).to_be_bytes()); + inputs.push( + prepared_for_store(format, *count, operation, provider.clone(), store.clone()).await?, + ); + } + Ok(inputs) +} +fn files(pack: &Prepared, budget: DiskBudget) -> Result { + Ok(NativeFiles::new( + Arc::new(tempfile::TempDir::new()?), + budget, + pack.store.clone(), + pack.descriptor.format, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )) +} +#[tokio::test] +async fn native_pack_cache_evicts_idle_files_for_disk_pressure_and_reuses_verified_downloads() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let inputs = packs(format, &[300, 301]).await?; + let size = inputs + .iter() + .map(|p| p.descriptor.pack.size + p.descriptor.index.size) + .max() + .ok_or("pair")?; + let budget = DiskBudget::new(size + 4096); + let files = files(&inputs[0], budget.clone())?; + drop(files.load(inputs[0].descriptor, Arc::new(())).await?); + drop(files.load(inputs[0].descriptor, Arc::new(())).await?); + assert_eq!(files.stats()?.downloaded_files, 1); + drop(files.load(inputs[1].descriptor, Arc::new(())).await?); + assert_eq!(files.stats()?.downloaded_files, 2); + assert_eq!(files.stats()?.cached_files, 1); + assert_eq!(files.stats()?.open_files, 1); + drop(files); + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await?; + } + Ok(()) +} +#[tokio::test] +async fn native_pack_slots_include_borrowed_evicted_files_and_refuse_capacity_without_deadlock() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let inputs = packs(format, &[12, 13, 14, 15, 16, 17, 18, 19, 20]).await?; + let files = files(&inputs[0], DiskBudget::new(64 << 20))?; + let mut borrowed = Vec::new(); + for pack in inputs.iter().take(OPEN_PACKS) { + borrowed.push(files.load(pack.descriptor, Arc::new(())).await?); + } + assert_eq!(files.stats()?.open_files, OPEN_PACKS); + assert_eq!(files.stats()?.cached_files, CACHED_PACKS); + assert!(matches!( + files.load(inputs[8].descriptor, Arc::new(())).await, + Err(NativeReadError::Capacity) + )); + borrowed.clear(); + let last = files.load(inputs[8].descriptor, Arc::new(())).await?; + assert_eq!(files.stats()?.downloaded_files, 9); + drop(last); + } + Ok(()) +} +#[tokio::test] +async fn native_pack_failure_never_enters_cache_and_file_slot_follows_native_cache_ownership() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let inputs = packs(format, &[16]).await?; + let descriptor = inputs[0].descriptor; + let files = files(&inputs[0], DiskBudget::new(64 << 20))?; + // Authenticated artifact bytes are intact; a false Git checksum must + // still fail pack/index binding before the cache remembers a file. + let mut wrong = descriptor; + wrong.git_checksum = match format { + ObjectFormat::Sha1 => crate::ObjectId::Sha1([9; 20]), + ObjectFormat::Sha256 => crate::ObjectId::Sha256([9; 32]), + }; + assert!(files.load(wrong, Arc::new(())).await.is_err()); + assert_eq!(files.stats()?.open_files, 0); + assert_eq!(files.stats()?.downloaded_files, 0); + let file = files.load(descriptor, Arc::new(())).await?; + let mut reader = + GitObjects::batch_owned(&file.cache.git_dir(), &files.native, file.cache.clone())?; + let expected = inputs[0].fixture.objects.values().next().ok_or("object")?.0; + reader.read_verified(expected, 1 << 20).await?; + files.cache.lock().map_err(|_| "cache")?.clear(); + drop(file); + assert_eq!(files.stats()?.cached_files, 0); + assert_eq!(files.stats()?.open_files, 1); + reader.finish().await?; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while files.stats().expect("stats").open_files != 0 { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/catalog/reader.rs b/crates/canopy-server/src/packs/catalog/reader.rs index 1d7d6407..f8755954 100644 --- a/crates/canopy-server/src/packs/catalog/reader.rs +++ b/crates/canopy-server/src/packs/catalog/reader.rs @@ -8,6 +8,7 @@ pub struct CatalogIndexes { ranges: RangeIndex, sources: Arc, inputs: super::super::sources::NativeInputIndex, + refs: super::super::ref_state::RefStateIndex, } impl CatalogIndexes { pub fn new(store: Arc, format: ObjectFormat) -> Self { @@ -15,6 +16,7 @@ impl CatalogIndexes { ranges: RangeIndex::new(Arc::clone(&store), format), sources: Arc::new(SourceIndex::new(Arc::clone(&store), format)), inputs: super::super::sources::NativeInputIndex::new(Arc::clone(&store), format), + refs: super::super::ref_state::RefStateIndex::new(Arc::clone(&store), format), store, } } @@ -27,6 +29,9 @@ impl CatalogIndexes { pub(in crate::packs) fn inputs(&self) -> &super::super::sources::NativeInputIndex { &self.inputs } + pub(in crate::packs) fn refs(&self) -> &super::super::ref_state::RefStateIndex { + &self.refs + } pub fn input_stats(&self) -> super::super::directory::index::ReadStats { self.inputs.stats() } @@ -101,6 +106,18 @@ impl CatalogReader { pub(in crate::packs) fn source_root(&self) -> Option { self.source_root } + pub(in crate::packs) fn source_changes( + &self, + before: Option, + after: Option, + ) -> Result< + super::super::directory::index::RangeChanges<'_, super::super::sources::SourceRecord>, + IndexError, + > { + self.indexes + .sources + .changes(before, self.source_root, after) + } pub async fn lookup( &self, oid: ObjectId, diff --git a/crates/canopy-server/src/packs/catalog/serving_fixture.rs b/crates/canopy-server/src/packs/catalog/serving_fixture.rs new file mode 100644 index 00000000..c7134f6f --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/serving_fixture.rs @@ -0,0 +1,287 @@ +//! Physically verified native inputs for serving tests. Catalog installation in +//! the caller is trusted injection, not qualification of the live publisher. +use super::*; +use crate::packs::{ + directory::DirectoryBuilder, + metadata::{ + PAGE_OBJECTS, + tests::{fixture_with_input, git, limits}, + }, + ref_state::{RefStateRecord, RefStateSnapshot, RefStateSnapshotRoot, RefStateTree}, + sources::{SourceIndex, SourceRecord}, + verification::{ + PhysicalVerifier, + physical::tests::{physical_limits, upload_fixture}, + }, +}; +use crate::{ObjectId, RefExpectation}; +use cellule_ltx::DiskBudget; +use object_store::ObjectStore; + +type Result = std::result::Result>; +pub(crate) struct BrowseFixture { + pub catalog: StoredCatalog, + pub refs: RefStateSnapshotRoot, + pub main: ObjectId, + pub side: ObjectId, + pub root: ObjectId, + pub previous: ObjectId, + pub tag: ObjectId, + pub tree: ObjectId, + pub wide: Option, + pub history: Vec, + pub edges: std::collections::BTreeMap>, + pub large: Vec, +} +pub(crate) fn operation(n: u64) -> [u8; 16] { + let mut id = *b"CANOPY0100000000"; + id[8..].copy_from_slice(&n.to_be_bytes()); + id +} +fn file(input: &mut Vec, mode: &str, name: &str, body: &[u8]) { + input.extend_from_slice(format!("M {mode} inline {name}\ndata {}\n", body.len()).as_bytes()); + input.extend_from_slice(body); + input.push(b'\n'); +} +fn commit(input: &mut Vec, branch: &str, mark: u32, parents: &[u32]) { + input.extend_from_slice(format!("commit refs/heads/{branch}\nmark :{mark}\ncommitter Browse Test {mark} +0000\ndata 7\nfixture\n").as_bytes()); + if let Some(first) = parents.first() { + input.extend_from_slice(format!("from :{first}\n").as_bytes()); + } + for parent in parents.iter().skip(1) { + input.extend_from_slice(format!("merge :{parent}\n").as_bytes()); + } +} +pub(crate) async fn prepare( + format: ObjectFormat, + provider: Arc, + repository: [u8; 16], + regular_files: usize, +) -> Result { + let mut input = Vec::new(); + commit(&mut input, "main", 1, &[]); + for n in 0..regular_files { + file( + &mut input, + "100644", + &format!("file-{n:04}"), + format!("original {n}\n").as_bytes(), + ); + } + file( + &mut input, + "100644", + "src/lib.rs", + b"pub fn original() {}\n", + ); + file(&mut input, "120000", "link", b"src/lib.rs"); + file(&mut input, "100755", "executable", b"#!/bin/sh\nexit 0\n"); + file(&mut input, "100644", "\"\\377name\"", b"raw name\n"); + file(&mut input, "100644", "\"literal[?]*\"", b"literal path\n"); + file(&mut input, "100644", "binary", b"a\0b\xff"); + let large = vec![b'x'; 256 * 1024 + 1]; + file(&mut input, "100644", "large", &large); + input.extend_from_slice( + format!("M 160000 {} submodule\n\n", "8".repeat(format.bytes() * 2)).as_bytes(), + ); + for mark in 2..=40 { + commit(&mut input, "main", mark, &[mark - 1]); + file( + &mut input, + "100644", + "file-0000", + format!("version {mark}\n").as_bytes(), + ); + input.push(b'\n'); + } + commit(&mut input, "side", 41, &[1]); + file(&mut input, "100644", "side-file", b"side\n"); + input.push(b'\n'); + commit(&mut input, "main", 42, &[40, 41]); + file(&mut input, "100644", "side-file", b"side\n"); + input.push(b'\n'); + if regular_files >= 600 { + // An actual 532-parent Git merge exercises ancestry continuation beyond + // one 512-edge page. Empty auxiliary commits share the root content. + for mark in 50..580 { + commit(&mut input, "aux", mark, &[1]); + input.push(b'\n'); + } + let mut parents = vec![1, 41]; + parents.extend(50..580); + commit(&mut input, "wide", 600, &parents); + input.push(b'\n'); + } + let fixture = fixture_with_input(format, input) + .await + .map_err(|e| e.to_string())?; + async fn oid(path: &std::path::Path, name: &str) -> Result { + let bytes = git(path, &["rev-parse", name], None) + .await + .map_err(|e| e.to_string())?; + Ok(ObjectId::from_hex(std::str::from_utf8(&bytes)?.trim())?) + } + let main = oid(fixture.root.path(), "main").await?; + let side = oid(fixture.root.path(), "side").await?; + let root = oid(fixture.root.path(), "main~40").await?; + let previous = oid(fixture.root.path(), "main~1").await?; + let tag = oid(fixture.root.path(), "metadata").await?; + let tree = oid(fixture.root.path(), "main^{tree}").await?; + let wide = if regular_files >= 600 { + Some(oid(fixture.root.path(), "wide").await?) + } else { + None + }; + let history = String::from_utf8( + git( + fixture.root.path(), + &["rev-list", "--first-parent", "main"], + None, + ) + .await + .map_err(|e| e.to_string())?, + )? + .lines() + .map(str::to_owned) + .collect(); + let edges = fixture + .objects + .iter() + .map(|(oid, (_, edges))| (*oid, edges.clone())) + .collect(); + let store = Arc::new(ArtifactStore::new(provider.clone(), repository)); + let native = upload_fixture(fixture, operation(100), provider, store.clone()) + .await + .map_err(|e| e.to_string())?; + let work = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut verifier = PhysicalVerifier::download( + work.path(), + budget.clone(), + &store, + native.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let mut segments = Vec::new(); + let mut remaining = native.descriptor.object_count; + while remaining != 0 { + let count = remaining.min(PAGE_OBJECTS as u32); + segments.push(verifier.inspect_next_shard(count).await?); + remaining -= count; + } + verifier + .finish() + .await? + .verify_segments(segments.iter().map(|s| s.descriptor()))?; + let mut directory = DirectoryBuilder::new( + work.path(), + budget.clone(), + repository, + operation(101), + format, + limits(), + )?; + let index = SourceIndex::new(store.clone(), format); + let mut sources = None; + for segment in &segments { + directory.add_segment(segment)?; + sources = Some( + index + .insert( + sources, + operation(102), + SourceRecord { + metadata: segment.clone().upload(&store).await?, + pack: native.descriptor.pack, + index: native.descriptor.index, + pack_object_count: native.descriptor.object_count, + }, + ) + .await?, + ); + } + let run = Arc::new(directory.seal()?).upload(&store).await?; + let ranges = RangeIndex::new(store.clone(), format); + let run_root = ranges.insert(None, operation(103), run).await?; + let mut directory = DirectorySnapshot::empty(repository, format); + directory.append(&ranges, run_root).await?; + let catalog = CatalogSnapshot { + directory: directory.upload(&store, operation(104)).await?, + sources, + } + .upload(&store, operation(105)) + .await?; + let refs = RefStateTree::new(store.clone(), format) + .build_sorted( + operation(106), + [ + RefStateRecord::new( + "refs/heads/main", + RefExpectation { + oid: Some(main), + version: 1, + }, + format, + )?, + RefStateRecord::new( + "refs/heads/side", + RefExpectation { + oid: Some(side), + version: 1, + }, + format, + )?, + ] + .into_iter() + .map(Ok), + ) + .await?; + let refs = RefStateSnapshotRoot::upload( + &store, + operation(107), + RefStateSnapshot { + repository, + format, + generation: 1, + default_branch: "refs/heads/main".into(), + root: refs, + }, + ) + .await?; + drop(segments); + tokio::time::timeout(std::time::Duration::from_secs(8), async { + while budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await?; + Ok(BrowseFixture { + catalog, + refs, + main, + side, + root, + previous, + tag, + tree, + wide, + history, + edges, + large, + }) +} + +/// Managed test-only Git invocation shared by native serving and HTTP fixtures. +/// Production commands use the admitted native service and physical owners. +pub(crate) async fn run_git( + path: &std::path::Path, + args: &[&str], + input: Option>, +) -> Result> { + git(path, args, input) + .await + .map_err(|error| error.to_string().into()) +} diff --git a/crates/canopy-server/src/packs/directory/index/changes.rs b/crates/canopy-server/src/packs/directory/index/changes.rs new file mode 100644 index 00000000..f47b8f50 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/changes.rs @@ -0,0 +1,200 @@ +//! Ordered additions/replacements between retained immutable roots. Equal +//! authenticated subtrees are skipped; no historical key set is materialized. +use super::*; + +#[derive(Clone)] +enum Item { + Node(NodeRef), + Record(R), +} +impl Item { + fn first(&self) -> R::Key { + match self { + Self::Node(n) => n.first_key.clone(), + Self::Record(r) => r.first_key(), + } + } + fn last(&self) -> R::Key { + match self { + Self::Node(n) => n.last_key.clone(), + Self::Record(r) => r.last_key(), + } + } +} + +/// Differences are defined by record first key and full record equality. +/// Deletions are omitted. A canceled/failed page poisons the cursor; restart +/// with the same root pair and the last *returned* record's first key. +pub struct RangeChanges<'a, R: IndexRecord> { + index: &'a RangeIndex, + before: Vec>, + after: Vec>, + resume: Option, + pending: Option, + initialized: bool, + poisoned: bool, +} +impl RangeIndex { + pub fn changes( + &self, + before: Option>, + after: Option>, + resume: Option, + ) -> Result, IndexError> { + for root in [&before, &after].into_iter().flatten() { + root.validate(self.format)?; + } + if resume.as_ref().is_some_and(|key| !key.valid(self.format)) { + return Err(IndexError::Integrity); + } + Ok(RangeChanges { + index: self, + before: before.into_iter().map(Item::Node).collect(), + after: after.into_iter().map(Item::Node).collect(), + resume, + pending: None, + initialized: false, + poisoned: false, + }) + } +} +impl RangeChanges<'_, R> { + async fn expand(index: &RangeIndex, stack: &mut Vec>) -> Result<(), IndexError> { + let Some(Item::Node(reference)) = stack.pop() else { + return Err(IndexError::Integrity); + }; + let node = index.load(reference).await?; + match &node.contents { + Contents::Children(v) => stack.extend(v.iter().rev().cloned().map(Item::Node)), + Contents::Runs(v) => stack.extend(v.iter().rev().cloned().map(Item::Record)), + } + // At most one path's unvisited siblings per level, including its leaf. + if stack.len() > (usize::from(R::MAX_HEIGHT) + 1) * R::FANOUT { + return Err(IndexError::Limit); + } + Ok(()) + } + async fn next_inner(&mut self) -> Result, IndexError> { + if !self.initialized { + self.initialized = true; + // Authenticate root context even when the entire pair is equal. + for stack in [&self.before, &self.after] { + if let Some(Item::Node(root)) = stack.last() { + self.index.validate_root(root.clone()).await?; + } + } + } + if let Some(record) = self.pending.take() { + return Ok(Some(record)); + } + loop { + let Some(new) = self.after.last() else { + return Ok(None); + }; + if self.resume.as_ref().is_some_and(|key| new.last() <= *key) { + self.after.pop(); + continue; + } + if let Some(old) = self.before.last() { + if let (Item::Node(a), Item::Node(b)) = (old, new) + && a == b + { + self.before.pop(); + self.after.pop(); + continue; + } + if old.last() < new.first() { + self.before.pop(); + continue; + } + if old.first() <= new.last() { + match (old, new) { + (Item::Node(a), Item::Node(b)) if a.height >= b.height => { + Self::expand(self.index, &mut self.before).await?; + continue; + } + (_, Item::Node(_)) => { + Self::expand(self.index, &mut self.after).await?; + continue; + } + (Item::Node(_), _) => { + Self::expand(self.index, &mut self.before).await?; + continue; + } + (Item::Record(a), Item::Record(b)) => { + // Range overlap is allowed between generations. Matching + // is by first key, rather than by enclosing interval. + if a.first_key() < b.first_key() { + self.before.pop(); + continue; + } + if a.first_key() == b.first_key() { + let equal = a == b; + self.before.pop(); + if equal { + self.after.pop(); + continue; + } + } + } + } + } + } + match self.after.last() { + Some(Item::Node(_)) => Self::expand(self.index, &mut self.after).await?, + Some(Item::Record(_)) => { + let Some(Item::Record(record)) = self.after.pop() else { + return Err(IndexError::Integrity); + }; + if self + .resume + .as_ref() + .is_none_or(|key| record.first_key() > *key) + { + return Ok(Some(record)); + } + } + None => return Ok(None), + } + } + } + /// Bound both record count and encoded descriptor bytes. Native input bytes + /// are separately admitted before downloads. Never advance over a record + /// excluded by the byte limit, including when a short page is returned. + pub async fn page(&mut self, count: usize, bytes: usize) -> Result, IndexError> { + if self.poisoned { + return Err(IndexError::Integrity); + } + if count == 0 || count > R::FANOUT || bytes == 0 || bytes > R::NODE_BYTES as usize { + return Err(IndexError::Limit); + } + self.poisoned = true; + let result = self.page_inner(count, bytes).await; + if result.is_ok() { + self.poisoned = false; + } + result + } + async fn page_inner(&mut self, count: usize, bytes: usize) -> Result, IndexError> { + let mut page = Vec::with_capacity(count); + let mut used = 0; + while page.len() < count { + let Some(record) = self.next_inner().await? else { + break; + }; + let mut encoder = BoundedEncoder::new(R::NODE_BYTES)?; + record.encode_record(&mut encoder)?; + let size = encoder.finish().len(); + if size > bytes - used { + if page.is_empty() { + return Err(IndexError::Limit); + } + self.pending = Some(record); + break; + } + used += size; + page.push(record); + } + Ok(page) + } +} diff --git a/crates/canopy-server/src/packs/directory/index/mod.rs b/crates/canopy-server/src/packs/directory/index/mod.rs index 7fe9cf56..7410100f 100644 --- a/crates/canopy-server/src/packs/directory/index/mod.rs +++ b/crates/canopy-server/src/packs/directory/index/mod.rs @@ -14,9 +14,11 @@ pub(in crate::packs) mod codec; pub(in crate::packs) mod record; pub use record::{IndexKey, IndexRecord}; mod bulk; +mod changes; mod cursor; mod rewrite; mod update; +pub use changes::RangeChanges; pub use cursor::RangeCursor; pub const FANOUT: usize = 128; diff --git a/crates/canopy-server/src/packs/metadata/tests.rs b/crates/canopy-server/src/packs/metadata/tests.rs index da9da7cb..ce668043 100644 --- a/crates/canopy-server/src/packs/metadata/tests.rs +++ b/crates/canopy-server/src/packs/metadata/tests.rs @@ -5,7 +5,11 @@ use tokio::{io::AsyncWriteExt, process::Command}; type Result = std::result::Result>; -async fn git(path: &Path, args: &[&str], input: Option>) -> Result> { +pub(in crate::packs) async fn git( + path: &Path, + args: &[&str], + input: Option>, +) -> Result> { let mut command = Command::new("git"); command .env_clear() @@ -43,6 +47,23 @@ pub(in crate::packs) struct Fixture { pub(in crate::packs) objects: BTreeMap)>, } pub(in crate::packs) async fn fixture(format: ObjectFormat, blobs: usize) -> Result { + // fast-import avoids one process per fixture object. The input is a test + // fixture; production verification streams native bodies into bounded SQL. + let mut input = b"commit refs/heads/main\ncommitter Metadata Test 1 +0000\ndata 7\nfixture\n".to_vec(); + for n in 0..blobs { + let body = format!("fixture body {n}\n"); + input.extend_from_slice( + format!("M 100644 inline file-{n}\ndata {}\n{body}", body.len()).as_bytes(), + ); + } + input.extend_from_slice(b"\n"); + fixture_with_input(format, input).await +} + +pub(in crate::packs) async fn fixture_with_input( + format: ObjectFormat, + input: Vec, +) -> Result { let root = tempfile::TempDir::new()?; git( root.path(), @@ -54,16 +75,6 @@ pub(in crate::packs) async fn fixture(format: ObjectFormat, blobs: usize) -> Res None, ) .await?; - // fast-import avoids one process per fixture object. The input is a test - // fixture; production verification streams native bodies into bounded SQL. - let mut input = b"commit refs/heads/main\ncommitter Metadata Test 1 +0000\ndata 7\nfixture\n".to_vec(); - for n in 0..blobs { - let body = format!("fixture body {n}\n"); - input.extend_from_slice( - format!("M 100644 inline file-{n}\ndata {}\n{body}", body.len()).as_bytes(), - ); - } - input.extend_from_slice(b"\n"); git(root.path(), &["fast-import", "--quiet"], Some(input)).await?; git( root.path(), diff --git a/crates/canopy-server/src/packs/publication/admission_receipt.rs b/crates/canopy-server/src/packs/publication/admission_receipt.rs new file mode 100644 index 00000000..04b86c8f --- /dev/null +++ b/crates/canopy-server/src/packs/publication/admission_receipt.rs @@ -0,0 +1,295 @@ +//! Bounded first-admission knowledge shared by staging and preparation. +//! These authenticated records grant no current custody or product access. +use super::{ + certificate::CertificateEnvelope, + recovery::{Stamp, phase::Recorded}, + sql::*, + *, +}; +use cellule_runtime::{ + CellClient, CellTarget, Committed, InvocationError, PendingMutation, Receipt, + primitives::sql::SqlCell, +}; +use std::marker::PhantomData; + +#[derive(Debug, thiserror::Error)] +pub enum AdmissionReceiptError { + #[error("initial admission receipt binding differs")] + Context, + #[error("initial admission receipt encoding failed")] + Codec(#[from] CodecError), + #[error("initial admission receipt capability failed")] + Capability(#[from] Error), + #[error("initial admission receipt query failed")] + Query(#[source] Box>>), +} + +/// Implemented only by the two private admission kinds. A column or purpose +/// cannot be chosen by an external caller. +pub(super) trait Admission: Clone + Send + 'static { + const DOMAIN: &'static [u8]; + const COLUMN: &'static str; + type Lease: Clone; + type Reply: WireValue; + fn grant(lease: Self::Lease) -> Self::Reply; + fn lease(result: &Recorded, request: &BeginRequest) -> Result; + fn token(lease: &Self::Lease) -> PreparationToken; +} + +#[derive(Clone)] +struct Record { + tenant: [u8; 16], + application: [u8; 16], + request: BeginRequest, + stamp: Stamp, + result: Recorded, + kind: PhantomData, +} +impl WireValue for Record { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + A::lease(&self.result, &self.request)?; + e.write_bytes(A::DOMAIN)?; + e.write_bytes(&self.tenant)?; + e.write_bytes(&self.application)?; + self.request.encode(e)?; + self.stamp.encode(e)?; + self.result.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != A::DOMAIN { + return Err(CodecError::Invalid("initial admission receipt purpose")); + } + let value = Self { + tenant: crate::packs::directory::index::codec::fixed(d)?, + application: crate::packs::directory::index::codec::fixed(d)?, + request: BeginRequest::decode(d)?, + stamp: Stamp::decode(d)?, + result: Recorded::decode(d)?, + kind: PhantomData, + }; + A::lease(&value.result, &value.request)?; + Ok(value) + } +} + +fn decode(bytes: &[u8], seed: &[u8; 32]) -> Result, AdmissionReceiptError> { + let mut d = BoundedDecoder::new(bytes, CERTIFICATE_BYTES)?; + let envelope = CertificateEnvelope::decode(&mut d)?; + d.finish()?; + if !envelope.authenticated(seed) { + return Err(AdmissionReceiptError::Context); + } + Ok(envelope.data()?) +} + +#[derive(Clone)] +pub(super) struct InitialAdmission { + target: CellTarget, + record: Record, +} +impl InitialAdmission { + pub(super) async fn load( + client: &CellClient, + target: &CellTarget, + operation: [u8; 16], + ) -> Result, AdmissionReceiptError> { + let sql = SqlCell::::new(client.clone(), target.clone())?; + let result = sql + .query( + None, + SqlBatch { + statements: vec![ + SqlStatement { + sql: format!( + "SELECT actor,request_digest,{} FROM pushes WHERE id=?1", + A::COLUMN + ), + parameters: vec![blob(operation)], + }, + SqlStatement { + sql: "SELECT push_cert_seed FROM repository_identity WHERE singleton=1" + .into(), + parameters: vec![], + }, + ], + }, + ) + .await + .map_err(|e| AdmissionReceiptError::Query(Box::new(e)))?; + let row = rows(&result.output)?.first().map(Vec::as_slice); + let Some([SqlValue::Text(actor), digest, value]) = row else { + return if row.is_none() { + Ok(None) + } else { + Err(AdmissionReceiptError::Context) + }; + }; + let bytes = match value { + SqlValue::Null => return Ok(None), + SqlValue::Blob(bytes) => bytes, + _ => return Err(AdmissionReceiptError::Context), + }; + let seed = attestation::seed( + result + .output + .get(1..) + .ok_or(AdmissionReceiptError::Context)?, + )?; + let record: Record = decode(bytes, &seed)?; + if record.request.actor != *actor + || record.request.request_digest != fixed::<32>(digest)? + || record.request.operation != operation + || record.tenant != *target.tenant().as_bytes() + || record.application != *target.application().as_bytes() + || crate::repository_target( + target.tenant(), + target.application(), + record.request.repository, + )? != *target + { + return Err(AdmissionReceiptError::Context); + } + Ok(Some(Self { + target: target.clone(), + record, + })) + } + pub(super) fn request(&self) -> &BeginRequest { + &self.record.request + } + pub(super) fn target(&self) -> &CellTarget { + &self.target + } + pub(super) fn lease(&self) -> A::Lease { + A::lease(&self.record.result, &self.record.request) + .expect("authenticated initial admission") + } + pub(super) fn receipt(&self) -> Receipt { + Receipt { + cell: self.target.cell_id(), + incarnation: A::token(&self.lease()).owner.incarnation, + // A Begin observing an already-bound attempt has its own receipt. + // The attempt sequence is not necessarily this command's sequence. + commit_sequence: self.record.result.sequence(), + } + } + pub(super) fn original( + &self, + evidence: &PendingMutation, + ) -> Result>, AdmissionReceiptError> { + if self.record.stamp != Stamp::of(evidence) { + return Ok(None); + } + if evidence.target() != &self.target || evidence.incarnation() != self.receipt().incarnation + { + return Err(AdmissionReceiptError::Context); + } + Ok(Some(self.record.result.committed(evidence)?)) + } +} + +/// The receipt is written after domain admission in the same SDK transaction. +/// Failure rolls back namespace allocation, custody and SDK knowledge together. +pub(super) fn save( + context: &CommandContext<'_, '_>, + request: &BeginRequest, + lease: A::Lease, +) -> cellule_runtime::Result<()> { + let evidence = context + .mutation_evidence() + .ok_or(Error::Command("initial admission evidence missing"))?; + if A::token(&lease).owner.incarnation != evidence.incarnation() { + return Err(Error::Command("initial admission incarnation differs")); + } + let mut output = BoundedEncoder::new(512)?; + A::grant(lease).encode(&mut output)?; + let record = Record:: { + tenant: *evidence.target().tenant().as_bytes(), + application: *evidence.target().application().as_bytes(), + request: request.clone(), + stamp: Stamp::of(&evidence), + result: Recorded::new(context.sequence(), false, output.finish())?, + kind: PhantomData, + }; + let seed = attestation::seed(&context.sql(&statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ))?)?; + let envelope = CertificateEnvelope::seal(&record, &seed)?; + let mut encoded = BoundedEncoder::new(CERTIFICATE_BYTES)?; + envelope.encode(&mut encoded)?; + let existing = context.sql(&statement( + &format!( + "SELECT actor,request_digest,{} FROM pushes WHERE id=?1", + A::COLUMN + ), + vec![blob(request.operation)], + ))?; + let saved = if let Some(row) = rows(&existing)?.first() { + let [SqlValue::Text(actor), digest, original] = row.as_slice() else { + return Err(Error::Command("invalid initial admission request row")); + }; + if *actor != request.actor || fixed::<32>(digest)? != request.request_digest { + return Err(Error::Command("initial admission request identity differs")); + } + if *original != SqlValue::Null { + return Ok(()); + } + statement( + &format!( + "UPDATE pushes SET {0}=?1 WHERE id=?2 AND {0} IS NULL AND response_id IS NULL", + A::COLUMN + ), + vec![blob(encoded.finish()), blob(request.operation)], + ) + } else { + statement( + &format!( + "INSERT INTO pushes(id,actor,request_digest,{}) VALUES(?1,?2,?3,?4)", + A::COLUMN + ), + vec![ + blob(request.operation), + SqlValue::Text(request.actor.clone()), + blob(request.request_digest), + blob(encoded.finish()), + ], + ) + }; + publish::changed(context.sql(&saved)?)?; + Ok(()) +} + +pub(super) fn restart_matches( + context: &CommandContext<'_, '_>, + check: &LeaseCheck, +) -> cellule_runtime::Result { + let sets = context.sql(&statement( + &format!( + "SELECT actor,request_digest,{} FROM pushes WHERE id=?1", + A::COLUMN + ), + vec![blob(check.token.operation)], + ))?; + let Some([SqlValue::Text(actor), digest, SqlValue::Blob(bytes)]) = + rows(&sets)?.first().map(Vec::as_slice) + else { + return Ok(false); + }; + if *actor != check.actor || fixed::<32>(digest)? != check.token.request_digest { + return Ok(false); + } + let seed = attestation::seed(&context.sql(&statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ))?)?; + let record: Record = decode(bytes, &seed) + .map_err(|_| Error::Command("initial admission receipt authentication failed"))?; + let evidence = context + .mutation_evidence() + .ok_or(Error::Command("Claim evidence missing"))?; + Ok(record.request.actor == check.actor + && A::token(&A::lease(&record.result, &record.request)?) == check.token + && record.tenant == *evidence.target().tenant().as_bytes() + && record.application == *evidence.target().application().as_bytes()) +} diff --git a/crates/canopy-server/src/packs/publication/base.rs b/crates/canopy-server/src/packs/publication/base.rs index df527c3e..cdcf7e01 100644 --- a/crates/canopy-server/src/packs/publication/base.rs +++ b/crates/canopy-server/src/packs/publication/base.rs @@ -14,12 +14,12 @@ use tokio::time::{Instant, timeout_at}; #[derive(Debug, thiserror::Error)] pub enum PreparationBaseError { + #[error("authoritative owner observation failed")] + Owner(#[source] Box), #[error("authoritative preparation query failed")] Query(#[source] Box>>), #[error("authoritative preparation frontier query failed")] Frontier(#[source] Box>>), - #[error("preparation renewal failed")] - Command(#[source] Box>), #[error("preparation catalog loading failed")] Catalog(#[from] IndexError), #[error("preparation has no active matching lease")] @@ -45,8 +45,9 @@ impl PreparationBaseResolver { indexes: Arc, files: Arc, minimum: Option, + authority: PreparationAuthority, ) -> Result { - let session = PreparationSession::open(client, target, check, minimum).await?; + let session = PreparationSession::open(client, target, check, minimum, authority).await?; Self::from_session(session, indexes, files).await } pub(super) async fn from_session( @@ -54,6 +55,7 @@ impl PreparationBaseResolver { indexes: Arc, files: Arc, ) -> Result { + session.check_owner().await?; let (lease, deadline) = session.live_lease()?; if indexes.store().repository() != lease.token.repository || indexes.sources().format() != lease.format @@ -69,6 +71,7 @@ impl PreparationBaseResolver { } else { None }; + session.check_owner().await?; session.live_lease()?; Ok(Self { session, @@ -127,6 +130,7 @@ impl PreparationBaseResolver { /// Select only facts read through the exact active attempt. The original /// floor, namespace, deadline and renewal fence are shared by all selections. pub(super) async fn select_current(&self) -> Result { + self.session.check_owner().await?; let (_, deadline) = self.live_lease()?; timeout_at(deadline, async { let started = Instant::now(); @@ -177,6 +181,7 @@ impl PreparationBaseResolver { None => None, } }; + self.session.check_owner().await?; self.live_lease()?; if Instant::now() >= deadline { return Err(PreparationBaseError::Inactive); @@ -192,12 +197,19 @@ impl PreparationBaseResolver { .await .map_err(|_| PreparationBaseError::Inactive)? } - pub async fn renew( + /// Preparation does not submit a command. The caller transfers this exact + /// renewal into the service-owned publication coordinator. + pub async fn ready_renew( &self, identity: MutationIdentity, lease_ms: u64, - ) -> Result<(), PreparationBaseError> { - self.session.renew(identity, lease_ms).await + ) -> Result { + Arc::new(self.session.clone()) + .ready_renew(identity, lease_ms) + .await + } + pub async fn restore_renewal(&self) -> Result { + Arc::new(self.session.clone()).restore_renewal().await } } impl BaseResolver for PreparationBaseResolver { diff --git a/crates/canopy-server/src/packs/publication/commands.rs b/crates/canopy-server/src/packs/publication/commands.rs index abc0aeeb..3ebfaacb 100644 --- a/crates/canopy-server/src/packs/publication/commands.rs +++ b/crates/canopy-server/src/packs/publication/commands.rs @@ -151,7 +151,7 @@ pub struct BeginPreparation; impl Command for BeginPreparation { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 11; - const CODEC_VERSION: u32 = 1; + const CODEC_VERSION: u32 = 2; type Input = BeginRequest; type Output = PreparationReply; fn execute( @@ -188,8 +188,10 @@ impl Command for BeginPreparation { } check_pin(context, &existing)?; let base = fact(context, input.repository, format, existing.generation)?; + let lease = grant(&existing, format, base, now)?; + super::preparation_receipt::save(context, &input, lease)?; return Ok(CommandResult::Success(PreparationReply::Granted(Box::new( - grant(&existing, format, base, now)?, + lease, )))); } if !quota(context, true)? { @@ -205,13 +207,15 @@ impl Command for BeginPreparation { insert_lease(context, new_token, Some(base.generation), expires)?; context.sql(&statement("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,?6,?7,?8,?9)",vec![blob(input.operation),SqlValue::Text(input.actor.clone()),blob(input.request_digest),blob(new_token.owner.incarnation.as_bytes()),blob(new_token.owner.epoch.to_be_bytes()),number(new_token.attempt)?,blob(new_token.artifact_operation),number(base.generation)?,SqlValue::Integer(expires)]))?; let row = Operation { - actor: input.actor, + actor: input.actor.clone(), token: new_token, generation: Some(base.generation), expires, }; + let lease = grant(&row, format, base, now)?; + super::preparation_receipt::save(context, &input, lease)?; Ok(CommandResult::Success(PreparationReply::Granted(Box::new( - grant(&row, format, base, now)?, + lease, )))) } } @@ -220,7 +224,7 @@ pub struct ClaimPreparation; impl Command for ClaimPreparation { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 12; - const CODEC_VERSION: u32 = 1; + const CODEC_VERSION: u32 = 3; type Input = LeaseRequest; type Output = PreparationReply; fn execute( @@ -238,7 +242,44 @@ impl Command for ClaimPreparation { return Ok(denied(PreparationDenial::Unauthorized)); }; let Some(existing) = load(context, check.token)? else { - return Ok(denied(PreparationDenial::Missing)); + if !super::preparation_receipt::restart_matches(context, &check)? + && !super::custody::restart_matches(context, &check, false)? + { + return Ok(denied(PreparationDenial::Missing)); + } + let begin = BeginRequest { + repository: check.token.repository, + operation: check.token.operation, + request_digest: check.token.request_digest, + actor: check.actor.clone(), + lease_ms: input.lease_ms, + }; + if !logical_available(context, &begin)? { + return Ok(denied(PreparationDenial::Conflict)); + } + if !quota(context, true)? { + return Ok(denied(PreparationDenial::Capacity)); + } + let now = now(context.now_ms())?; + let expires = expiry(now, input.lease_ms)?; + let base = fact(context, check.token.repository, format, None)?; + let next = token( + context, + check.token.repository, + check.token.operation, + check.token.request_digest, + )?; + insert_lease(context, next, Some(base.generation), expires)?; + context.sql(&statement("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,?6,?7,?8,?9)", vec![blob(next.operation),SqlValue::Text(check.actor.clone()),blob(next.request_digest),blob(next.owner.incarnation.as_bytes()),blob(next.owner.epoch.to_be_bytes()),number(next.attempt)?,blob(next.artifact_operation),number(base.generation)?,SqlValue::Integer(expires)]))?; + let row = Operation { + actor: check.actor, + token: next, + generation: Some(base.generation), + expires, + }; + return Ok(CommandResult::Success(PreparationReply::Granted(Box::new( + grant(&row, format, base, now)?, + )))); }; if !matched(&existing, &check) { return Ok(denied(PreparationDenial::Stale)); @@ -453,7 +494,7 @@ impl Query for CheckPreparationFrontier { } pub struct ReapPreparation; -pub(super) const REAP_GENERATIONS: &str = "DELETE FROM catalog_generations WHERE generation IN (SELECT g.generation FROM catalog_generations g WHERE g.generation>0 AND g.generation<(SELECT generation FROM catalog_state WHERE singleton=1) AND g.generation, } +#[cfg(test)] +impl ReadyCatalogPush { + pub(in crate::packs::publication) fn evidence_for_test( + &self, + ) -> cellule_runtime::PendingMutation { + self.command.evidence().clone() + } +} #[derive(Clone)] enum PushPreparation { Catalog(Arc), @@ -233,8 +247,17 @@ struct ReadContext { target: CellTarget, request: BeginRequest, } +#[derive(Clone, Copy, PartialEq, Eq, Hash)] +enum JobKind { + Publication, + CustodyStop, + ServingRelease, + ServingCommand, + ServingStop, +} struct Job { operation: [u8; 16], + kind: JobKind, actor: String, // Removed before terminal notification; tickets never retain command // payloads or local inventory after the admission charge is released. @@ -246,6 +269,7 @@ struct Job { status: watch::Sender, read: ReadContext, admitted: Instant, + budget: std::sync::Mutex>, } struct Work { job: Arc, @@ -326,7 +350,7 @@ impl ClassQueue { } #[derive(Default)] struct State { - jobs: HashMap<[u8; 16], Arc>, + jobs: HashMap<([u8; 16], JobKind), Arc>, actors: HashMap, queue: ClassQueue, counts: [usize; 2], @@ -337,9 +361,11 @@ struct State { struct Inner { target: CellTarget, limits: PublicationLimits, + budget: PublicationBudget, state: Mutex, drained: Notify, changed: Notify, + serving_drain: std::sync::Mutex>, #[cfg(test)] gate: Mutex>, #[cfg(test)] @@ -379,18 +405,26 @@ pub struct PublicationStats { pub maintenance: usize, } impl PublicationCoordinator { + pub(in crate::packs::publication) fn target(&self) -> &CellTarget { + &self.inner.target + } + /// Every repository dispatcher on a node must receive the same budget. + /// Repository limits remain additional caps, not independent node shares. pub fn new( target: CellTarget, limits: PublicationLimits, + budget: PublicationBudget, ) -> Result { limits.validate()?; Ok(Self { inner: Arc::new(Inner { target, limits, + budget, state: Mutex::new(State::default()), drained: Notify::new(), changed: Notify::new(), + serving_drain: std::sync::Mutex::new(None), #[cfg(test)] gate: Mutex::new(None), #[cfg(test)] @@ -436,7 +470,8 @@ impl PublicationCoordinator { let class = ready.class(); let reservation = ready.reservation(); let at = class.index(); - let (client, target, check) = ready.capability(); + let (client, target, request) = ready.context(); + let kind = ready.job_kind(); let limits = self.inner.limits; let (operation_limit, byte_limit) = match class { PublicationClass::Foreground => ( @@ -451,17 +486,19 @@ impl PublicationCoordinator { }; let reason = if target != &self.inner.target { Some(PublicationScheduleError::Foreign) - } else if state.closed { + } else if state.closed || !self.inner.drain_allows(&ready) { Some(PublicationScheduleError::Closed) - } else if state.jobs.contains_key(&check.token.operation) { + } else if state.jobs.contains_key(&(request.operation, kind)) { Some(PublicationScheduleError::Duplicate) } else if state.counts[at] >= operation_limit || state .actors - .get(&check.actor) + .get(&request.actor) .map_or(0, |counts| counts[at]) >= limits.per_actor - || state.bytes[at] > byte_limit - reservation + || byte_limit + .checked_sub(reservation) + .is_none_or(|remaining| state.bytes[at] > remaining) { Some(PublicationScheduleError::Capacity) } else { @@ -470,20 +507,23 @@ impl PublicationCoordinator { if let Some(reason) = reason { return Err(Box::new(PublicationAdmissionFailure { reason, ready })); } + let budget = match self + .inner + .budget + .reserve(class, &request.actor, reservation) + { + Ok(permit) => permit, + Err(reason) => return Err(Box::new(PublicationAdmissionFailure { reason, ready })), + }; let read = ReadContext { client: client.clone(), target: target.clone(), - request: BeginRequest { - repository: check.token.repository, - operation: check.token.operation, - request_digest: check.token.request_digest, - actor: check.actor.clone(), - lease_ms: DEFAULT_LEASE_MS, - }, + request: request.clone(), }; let job = Arc::new(Job { - operation: check.token.operation, - actor: check.actor.clone(), + operation: request.operation, + kind, + actor: request.actor.clone(), class, reservation, policy_page: ready.is_policy_page(), @@ -497,11 +537,14 @@ impl PublicationCoordinator { .0, read, admitted: Instant::now(), + budget: std::sync::Mutex::new(Some(budget)), }); state.actors.entry(job.actor.clone()).or_default()[at] += 1; state.counts[at] += 1; state.bytes[at] += job.reservation; - state.jobs.insert(job.operation, Arc::clone(&job)); + state + .jobs + .insert((job.operation, job.kind), Arc::clone(&job)); if !held { enqueue(state, &job, false); self.start(state); @@ -525,7 +568,7 @@ impl PublicationCoordinator { let mut state = self.inner.state.lock().await; if !state .jobs - .get(&ticket.job.operation) + .get(&(ticket.job.operation, ticket.job.kind)) .is_some_and(|job| Arc::ptr_eq(job, &ticket.job)) || !matches!(*ticket.job.status.borrow(), PublicationState::Uncertain(_)) { @@ -549,6 +592,14 @@ impl PublicationCoordinator { pub(in crate::packs::publication) async fn recover_terminal_releases( &self, ) -> Result { + self.recover_retirements(false).await + } + pub(in crate::packs::publication) async fn recover_custody_stops( + &self, + ) -> Result { + self.recover_retirements(true).await + } + async fn recover_retirements(&self, custody: bool) -> Result { let jobs: Vec<_> = { let state = self.inner.state.lock().await; state @@ -565,10 +616,13 @@ impl PublicationCoordinator { // retain a command-body copy, replace an identity or retry compaction. let mut recovered = 0; for job in jobs { - let release = matches!( - &*job.ready.lock().await, - Some(ReadyPublication::TerminalRelease(_)) - ); + let ready = job.ready.lock().await; + let release = if custody { + matches!(&*ready, Some(ReadyPublication::CustodyStop(_))) + } else { + matches!(&*ready, Some(ReadyPublication::TerminalRelease(_))) + }; + drop(ready); if !release { continue; } @@ -596,6 +650,20 @@ impl PublicationCoordinator { wake.as_mut().enable(); { let mut state = self.inner.state.lock().await; + // An eviction owner must submit its remaining exact releases + // before global closure. Do not strand that owner's admission. + if !state.closed + && self + .inner + .serving_drain + .lock() + .expect("serving drain admission") + .is_some() + { + drop(state); + wake.await; + continue; + } state.closed = true; if !state.worker { return state @@ -611,15 +679,63 @@ impl PublicationCoordinator { wake.await; } } + /// Eviction closes only an already empty queue. Busy/uncertain owners keep + /// their admission and may resume discovery without replacing commands. + pub(crate) async fn close_if_idle(&self) -> bool { + let mut state = self.inner.state.lock().await; + if state.worker + || !state.jobs.is_empty() + || self + .inner + .serving_drain + .lock() + .expect("serving drain admission") + .is_some() + { + return false; + } + state.closed = true; + true + } /// Service-internal lookup after its caller loses a ticket. This is not an /// externally authorized product query; use completed-request replay there. pub async fn pending(&self, operation: [u8; 16]) -> Option { + self.pending_kind(operation, JobKind::Publication).await + } + /// Retirement has a separate bounded key kind, never a fabricated operation. + pub async fn pending_custody_stop(&self, operation: [u8; 16]) -> Option { + self.pending_custody_stop_for(CustodyPurpose::Creating, operation) + .await + } + pub(in crate::packs::publication) async fn pending_custody_stop_for( + &self, + purpose: CustodyPurpose, + operation: [u8; 16], + ) -> Option { + self.pending_kind( + operation, + if purpose == CustodyPurpose::Serving { + JobKind::ServingStop + } else { + JobKind::CustodyStop + }, + ) + .await + } + pub async fn pending_serving_command(&self, reader: [u8; 16]) -> Option { + self.pending_kind(reader, JobKind::ServingCommand).await + } + /// Read-retention release cannot collide with a creating request's ID. + pub async fn pending_serving_release(&self, reader: [u8; 16]) -> Option { + self.pending_kind(reader, JobKind::ServingRelease).await + } + async fn pending_kind(&self, operation: [u8; 16], kind: JobKind) -> Option { self.inner .state .lock() .await .jobs - .get(&operation) + .get(&(operation, kind)) .map(|job| PublicationTicket { inner: Arc::clone(&self.inner), job: Arc::clone(job), @@ -666,7 +782,7 @@ impl PublicationCoordinator { (release, start) } #[cfg(test)] - pub(super) fn fault_for_test(&self, fault: u8) { + pub(crate) fn fault_for_test(&self, fault: u8) { self.inner .fault .store(fault, std::sync::atomic::Ordering::Release); @@ -850,7 +966,10 @@ fn enqueue(state: &mut State, job: &Arc, recover: bool) { ); } fn release(state: &mut State, job: &Job) { - state.jobs.remove(&job.operation); + // Both callers drop retained proof/body ownership before making either + // the repository or node reservation reusable. Ticket DTOs may survive. + job.budget.lock().expect("publication budget permit").take(); + state.jobs.remove(&(job.operation, job.kind)); let count = state .actors .get_mut(&job.actor) @@ -941,6 +1060,7 @@ async fn run(inner: Arc) { } } async fn dispatch(inner: Arc, work: Work) -> DispatchResult { + let _dispatch = inner.budget.dispatch(work.job.class, &work.job.actor).await; #[cfg(test)] { // Do not hold the hook's mutex across a wait: other dispatched jobs @@ -983,9 +1103,51 @@ async fn finish(inner: &Inner, job: &Job, outcome: DispatchResult) { .send_replace(PublicationState::Uncertain(Arc::new(outcome.unwrap_err()))); } else { // Drop large resources before making their admission reusable. - job.ready.lock().await.take(); + let retained = job.ready.lock().await.take(); + let released = if matches!(&outcome, Ok(PublicationOutcome::ServingRelease(value)) if value.output == ServingReleaseReply::Released) + { + retained + .as_ref() + .and_then(ReadyPublication::serving_release_token) + } else { + None + }; + drop(retained); let mut state = inner.state.lock().await; release(&mut state, job); + if let Some(token) = released { + inner.observe_serving_release(token); + } + // A retired original may itself occupy a foreground uncertainty slot. + // Resume only that exact preparation evidence, never unrelated work. + if let Ok(PublicationOutcome::CustodyStop(value)) = &outcome + && value.stop.is_some() + && let Some(original) = state + .jobs + .get(&( + job.operation, + if value.purpose == CustodyPurpose::Serving { + JobKind::ServingCommand + } else { + JobKind::Publication + }, + )) + .cloned() + && matches!(*original.status.borrow(), PublicationState::Uncertain(_)) + { + let matches = original + .ready + .lock() + .await + .as_ref() + .and_then(ReadyPublication::custody_original) + .is_some_and(|evidence| *evidence == value.original); + if matches { + original.status.send_replace(PublicationState::Queued); + enqueue(&mut state, &original, true); + inner.changed.notify_one(); + } + } job.status .send_replace(PublicationState::Finished(outcome.map_err(Arc::new))); } diff --git a/crates/canopy-server/src/packs/publication/coordinator/budget.rs b/crates/canopy-server/src/packs/publication/coordinator/budget.rs new file mode 100644 index 00000000..9ed87f5c --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator/budget.rs @@ -0,0 +1,202 @@ +//! One shared budget for every repository dispatcher on a node. Command credits +//! cover retained originals through uncertainty; transport slots cover dispatch. +use super::*; +use std::sync::Mutex as LedgerMutex; +use tokio::sync::{OwnedSemaphorePermit, Semaphore}; + +#[derive(Clone)] +pub struct PublicationBudget { + inner: Arc, +} +struct BudgetInner { + limits: PublicationLimits, + ledger: LedgerMutex, + dispatch: [Arc; 2], +} +#[derive(Default)] +struct Ledger { + counts: [usize; 2], + bytes: [u64; 2], + actors: HashMap, + closed: bool, +} +struct ActorBudget { + counts: [usize; 2], + dispatch: [Arc; 2], +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PublicationBudgetStats { + pub foreground: usize, + pub maintenance: usize, + pub accounts: usize, + pub command_bytes: u64, + pub foreground_dispatch: usize, + pub maintenance_dispatch: usize, + pub closed: bool, +} +impl PublicationBudget { + /// Reuse the dispatcher profile; foreground_burst remains a repository + /// scheduling setting. This budget reserves independent class shares. + pub fn new(limits: PublicationLimits) -> Result { + limits.validate()?; + if limits.maintenance_operations < 2 + || limits.maintenance_in_flight < 2 + || limits.in_flight - limits.maintenance_in_flight < 2 + { + return Err(PublicationScheduleError::InvalidLimits); + } + Ok(Self { + inner: Arc::new(BudgetInner { + limits, + ledger: LedgerMutex::new(Ledger::default()), + dispatch: [ + Arc::new(Semaphore::new( + limits.in_flight - limits.maintenance_in_flight, + )), + Arc::new(Semaphore::new(limits.maintenance_in_flight)), + ], + }), + }) + } + /// Stop new reservations without cancelling admitted dispatch or recovery. + pub fn close(&self) { + self.inner.ledger.lock().expect("publication budget").closed = true; + } + pub fn stats(&self) -> PublicationBudgetStats { + let ledger = self.inner.ledger.lock().expect("publication budget"); + let limits = self.inner.limits; + PublicationBudgetStats { + foreground: ledger.counts[0], + maintenance: ledger.counts[1], + accounts: ledger.actors.len(), + command_bytes: ledger.bytes.iter().sum(), + foreground_dispatch: limits.in_flight + - limits.maintenance_in_flight + - self.inner.dispatch[0].available_permits(), + maintenance_dispatch: limits.maintenance_in_flight + - self.inner.dispatch[1].available_permits(), + closed: ledger.closed, + } + } + pub(super) fn reserve( + &self, + class: PublicationClass, + actor: &str, + bytes: u64, + ) -> Result { + let mut ledger = self.inner.ledger.lock().expect("publication budget"); + if ledger.closed { + return Err(PublicationScheduleError::Closed); + } + let at = class.index(); + let limits = self.inner.limits; + let (operations, ceiling, actor_limit) = match class { + PublicationClass::Foreground => ( + limits.operations - limits.maintenance_operations, + limits.command_bytes + - limits.maintenance_operations as u64 * MAINTENANCE_RESERVATION, + limits.per_actor, + ), + PublicationClass::Maintenance => ( + limits.maintenance_operations, + limits.maintenance_operations as u64 * MAINTENANCE_RESERVATION, + limits.per_actor.min(limits.maintenance_operations / 2), + ), + }; + if bytes == 0 + || ledger.counts[at] >= operations + || ledger + .actors + .get(actor) + .map_or(0, |account| account.counts[at]) + >= actor_limit + || ceiling + .checked_sub(bytes) + .is_none_or(|remaining| ledger.bytes[at] > remaining) + { + return Err(PublicationScheduleError::Capacity); + } + ledger.counts[at] += 1; + ledger.bytes[at] += bytes; + ledger + .actors + .entry(actor.to_owned()) + .or_insert_with(|| ActorBudget { + counts: [0; 2], + dispatch: [ + Arc::new(Semaphore::new( + limits + .per_actor + .min((limits.in_flight - limits.maintenance_in_flight) / 2), + )), + Arc::new(Semaphore::new( + limits.per_actor.min(limits.maintenance_in_flight / 2), + )), + ], + }) + .counts[at] += 1; + Ok(BudgetPermit { + inner: Arc::clone(&self.inner), + class, + actor: actor.to_owned(), + bytes, + }) + } + pub(super) async fn dispatch(&self, class: PublicationClass, actor: &str) -> DispatchPermit { + // Semaphores are never closed: closing admission must retain exact + // recovery after shutdown, including jobs already waiting for a slot. + let actor_dispatch = { + let ledger = self.inner.ledger.lock().expect("publication budget"); + Arc::clone( + &ledger + .actors + .get(actor) + .expect("admitted publication account") + .dispatch[class.index()], + ) + }; + // Account waiters must not occupy node slots while waiting for their + // account share, including when one account spans many repositories. + let actor = actor_dispatch + .acquire_owned() + .await + .expect("owned publication account dispatch budget"); + let class = Arc::clone(&self.inner.dispatch[class.index()]) + .acquire_owned() + .await + .expect("owned publication dispatch budget"); + DispatchPermit { + _actor: actor, + _class: class, + } + } +} +pub(super) struct DispatchPermit { + _actor: OwnedSemaphorePermit, + _class: OwnedSemaphorePermit, +} +pub(super) struct BudgetPermit { + inner: Arc, + class: PublicationClass, + actor: String, + bytes: u64, +} +impl Drop for BudgetPermit { + fn drop(&mut self) { + let mut ledger = self.inner.ledger.lock().expect("publication budget"); + let at = self.class.index(); + ledger.counts[at] -= 1; + ledger.bytes[at] -= self.bytes; + let account = ledger + .actors + .get_mut(&self.actor) + .expect("admitted publication account"); + account.counts[at] -= 1; + if account.counts == [0; 2] { + ledger.actors.remove(&self.actor); + } + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/publication/coordinator/budget/tests.rs b/crates/canopy-server/src/packs/publication/coordinator/budget/tests.rs new file mode 100644 index 00000000..5d4f1bb4 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator/budget/tests.rs @@ -0,0 +1,264 @@ +use super::*; +use tokio::time::{Duration, timeout}; + +fn limits() -> PublicationLimits { + PublicationLimits { + operations: 8, + per_actor: 2, + command_bytes: 3 * COMMAND_RESERVATION + 4 * MAINTENANCE_RESERVATION, + in_flight: 4, + maintenance_operations: 4, + maintenance_in_flight: 2, + foreground_burst: 3, + } +} + +#[test] +fn aggregate_reservations_keep_class_and_account_headroom() { + let budget = PublicationBudget::new(limits()).unwrap(); + let other_repository = budget.clone(); + let foreground = PublicationClass::Foreground; + let maintenance = PublicationClass::Maintenance; + let a = budget + .reserve(foreground, "a", COMMAND_RESERVATION) + .unwrap(); + let b = other_repository + .reserve(foreground, "a", COMMAND_RESERVATION) + .unwrap(); + assert!(matches!( + budget.reserve(foreground, "a", 1), + Err(PublicationScheduleError::Capacity) + )); + let c = budget + .reserve(foreground, "b", COMMAND_RESERVATION) + .unwrap(); + assert!(matches!( + other_repository.reserve(foreground, "b", 1), + Err(PublicationScheduleError::Capacity) + )); + let m1 = budget + .reserve(maintenance, "a", MAINTENANCE_RESERVATION) + .unwrap(); + let m2 = other_repository + .reserve(maintenance, "a", MAINTENANCE_RESERVATION) + .unwrap(); + assert!(matches!( + budget.reserve(maintenance, "a", 1), + Err(PublicationScheduleError::Capacity) + )); + let m3 = budget + .reserve(maintenance, "b", MAINTENANCE_RESERVATION) + .unwrap(); + let m4 = budget + .reserve(maintenance, "c", MAINTENANCE_RESERVATION) + .unwrap(); + assert!(matches!( + budget.reserve(maintenance, "d", 1), + Err(PublicationScheduleError::Capacity) + )); + let stats = budget.stats(); + assert_eq!( + (stats.foreground, stats.maintenance, stats.accounts), + (3, 4, 3) + ); + assert_eq!(stats.command_bytes, limits().command_bytes); + drop((a, b, c)); + let foreground_again = budget + .reserve(foreground, "d", COMMAND_RESERVATION) + .unwrap(); + drop((foreground_again, m1, m2, m3, m4)); + let stats = budget.stats(); + assert_eq!( + ( + stats.foreground, + stats.maintenance, + stats.accounts, + stats.command_bytes + ), + (0, 0, 0, 0) + ); +} + +#[tokio::test] +async fn account_waiters_leave_node_slots_for_other_accounts_and_maintenance() { + let budget = PublicationBudget::new(limits()).unwrap(); + let mut reservations = Vec::new(); + for class in [PublicationClass::Foreground, PublicationClass::Maintenance] { + for actor in ["a", "a", "b", "c"] { + reservations.push(budget.reserve(class, actor, 1).unwrap()); + } + } + let foreground = PublicationClass::Foreground; + let maintenance = PublicationClass::Maintenance; + let a = budget.dispatch(foreground, "a").await; + let a_waiter = budget.dispatch(foreground, "a"); + tokio::pin!(a_waiter); + assert!( + timeout(Duration::from_millis(20), &mut a_waiter) + .await + .is_err() + ); + // If the account waiter acquired the node gate first, this would hang. + let b = timeout(Duration::from_secs(1), budget.dispatch(foreground, "b")) + .await + .unwrap(); + let c_waiter = budget.dispatch(foreground, "c"); + tokio::pin!(c_waiter); + assert!( + timeout(Duration::from_millis(20), &mut c_waiter) + .await + .is_err() + ); + let m_a = timeout(Duration::from_secs(1), budget.dispatch(maintenance, "a")) + .await + .unwrap(); + let m_a_waiter = budget.dispatch(maintenance, "a"); + tokio::pin!(m_a_waiter); + assert!( + timeout(Duration::from_millis(20), &mut m_a_waiter) + .await + .is_err() + ); + let m_b = timeout(Duration::from_secs(1), budget.dispatch(maintenance, "b")) + .await + .unwrap(); + assert_eq!( + ( + budget.stats().foreground_dispatch, + budget.stats().maintenance_dispatch + ), + (2, 2) + ); + budget.close(); + assert!(matches!( + budget.reserve(foreground, "d", 1), + Err(PublicationScheduleError::Closed) + )); + drop(a); + // c already entered the node FIFO while a's second job waited only on + // its account gate. c must receive the newly available node slot first. + let c = timeout(Duration::from_secs(1), &mut c_waiter) + .await + .unwrap(); + assert!( + timeout(Duration::from_millis(20), &mut a_waiter) + .await + .is_err() + ); + drop(b); + let a_again = timeout(Duration::from_secs(1), &mut a_waiter) + .await + .unwrap(); + drop(m_a); + let m_a_again = timeout(Duration::from_secs(1), &mut m_a_waiter) + .await + .unwrap(); + drop((a_again, c, m_a_again, m_b)); + assert_eq!( + ( + budget.stats().foreground_dispatch, + budget.stats().maintenance_dispatch + ), + (0, 0) + ); + drop(reservations); + assert_eq!(budget.stats().accounts, 0); +} + +#[tokio::test] +async fn canceling_dispatch_wait_keeps_command_credit_and_releases_partial_gates() { + let budget = PublicationBudget::new(limits()).unwrap(); + let foreground = PublicationClass::Foreground; + let mut reservations = Vec::new(); + for actor in ["a", "b", "c"] { + reservations.push(budget.reserve(foreground, actor, 1).unwrap()); + } + let a = budget.dispatch(foreground, "a").await; + let b = budget.dispatch(foreground, "b").await; + // c obtains its account gate, then waits for the global gate. Dropping + // that future must give c back its account share for exact recovery. + assert!( + timeout(Duration::from_millis(20), budget.dispatch(foreground, "c")) + .await + .is_err() + ); + assert_eq!(budget.stats().foreground, 3); + assert_eq!(budget.stats().foreground_dispatch, 2); + drop(a); + let c = timeout(Duration::from_secs(1), budget.dispatch(foreground, "c")) + .await + .unwrap(); + drop((b, c, reservations)); + assert_eq!( + ( + budget.stats().foreground, + budget.stats().foreground_dispatch, + budget.stats().accounts + ), + (0, 0, 0) + ); +} + +#[test] +fn invalid_profiles_and_oversized_credits_cannot_wrap_or_consume_reservations() { + for invalid in [ + PublicationLimits { + maintenance_operations: 1, + ..limits() + }, + PublicationLimits { + maintenance_in_flight: 1, + ..limits() + }, + PublicationLimits { + in_flight: 3, + ..limits() + }, + PublicationLimits { + in_flight: 0, + ..limits() + }, + PublicationLimits { + command_bytes: 0, + ..limits() + }, + ] { + assert!(matches!( + PublicationBudget::new(invalid), + Err(PublicationScheduleError::InvalidLimits) + )); + } + let budget = PublicationBudget::new(limits()).unwrap(); + for bytes in [0, u64::MAX, limits().command_bytes] { + assert!(matches!( + budget.reserve(PublicationClass::Foreground, "a", bytes), + Err(PublicationScheduleError::Capacity) + )); + } + assert_eq!( + (budget.stats().accounts, budget.stats().command_bytes), + (0, 0) + ); + let budget = PublicationBudget::new(PublicationLimits { + command_bytes: u64::MAX, + ..limits() + }) + .unwrap(); + let bytes = u64::MAX - 4 * MAINTENANCE_RESERVATION; + let foreground = budget + .reserve(PublicationClass::Foreground, "a", bytes) + .unwrap(); + assert!(matches!( + budget.reserve(PublicationClass::Foreground, "b", 1), + Err(PublicationScheduleError::Capacity) + )); + let maintenance = budget + .reserve(PublicationClass::Maintenance, "b", MAINTENANCE_RESERVATION) + .unwrap(); + assert_eq!( + budget.stats().command_bytes, + bytes + MAINTENANCE_RESERVATION + ); + drop((foreground, maintenance)); + assert_eq!(budget.stats().command_bytes, 0); +} diff --git a/crates/canopy-server/src/packs/publication/coordinator/initialization.rs b/crates/canopy-server/src/packs/publication/coordinator/initialization.rs new file mode 100644 index 00000000..bd229b2a --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator/initialization.rs @@ -0,0 +1,102 @@ +//! Initialization reuses exact registered recovery and original live custody. +use super::*; +use canopy_object_storage::artifact::ArtifactStore; + +/// Private empty-catalog proof and exact original SDK command. Registration +/// precedes every dispatch; unknown registration cannot authorize execution. +#[must_use] +pub struct ReadyInitialization { + owner: Arc, + command: PreparedCommand, +} +impl PreparedCatalog { + pub async fn ready_initialization( + self: &Arc, + identity: MutationIdentity, + ) -> Result { + let proof = self.empty_ref_initialization().await?; + let (client, target, _) = self.base.capability(); + self.ensure_live()?; + let command = client + .prepare_command::(target, identity, proof) + .await + .map_err(|error| InitializationPreparationError::Command(Box::new(error)))?; + self.ensure_live()?; + Ok(ReadyInitialization { + owner: Arc::clone(self), + command, + }) + } +} +impl ReadyInitialization { + pub async fn persist_recovery( + &self, + store: &ArtifactStore, + identity: MutationIdentity, + ) -> Result { + super::super::recovery::persist( + &self.owner.base.session, + &self.command, + super::super::recovery::Kind::Initialization, + store, + identity, + 0, + ) + .await + } + /// Reuses the account-fair publication queue and its exact live-session + /// binding. A decoded recovery record cannot substitute for this owner. + pub fn bind_recovery( + self, + registered: RegisteredRootRecovery, + store: &ArtifactStore, + ) -> Result>> { + if !self.matches(®istered, store) { + return Err(Box::new(RecoveryBindingFailure { + original: self, + registered, + })); + } + Ok(ReadyBoundRecovery::new( + PushPreparation::Catalog(self.owner), + None, + false, + registered, + store, + )) + } + fn matches(&self, registered: &RegisteredRootRecovery, store: &ArtifactStore) -> bool { + registered.matches_original( + super::super::recovery::Kind::Initialization, + self.command.evidence(), + None, + &self.owner.base.session, + store, + ) + } + /// Repository startup already owns its bounded transition admission. Keep + /// this exact original session through its registered command and all I/O. + pub(crate) async fn complete( + self, + registered: &RegisteredRootRecovery, + store: &ArtifactStore, + ) -> Result, PublicationError> { + if !self.matches(registered, store) { + return Err(PublicationError::Recovery { + evidence: Box::new(self.command.evidence().clone()), + source: Box::new(RootRecoveryError::Context), + }); + } + let client = self.owner.base.capability().0; + let result = registered + .dispatch_initialization( + client, + store, + &self.owner.base.session.authority, + Some(&self.owner.base.session), + ) + .await; + drop(self.owner); + result + } +} diff --git a/crates/canopy-server/src/packs/publication/coordinator/preparation.rs b/crates/canopy-server/src/packs/publication/coordinator/preparation.rs index 9ae78618..74454485 100644 --- a/crates/canopy-server/src/packs/publication/coordinator/preparation.rs +++ b/crates/canopy-server/src/packs/publication/coordinator/preparation.rs @@ -1,8 +1,9 @@ //! Bound attempt commands share exact publication ownership and admission. +use super::super::custody::OwnedCustody; use super::*; const INLINE_BYTES: u32 = 4096; -pub(super) const RESERVATION: u64 = 2 * INLINE_BYTES as u64; +pub(super) const RESERVATION: u64 = super::super::custody::RESERVATION; #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum PreparationCommandKind { @@ -11,19 +12,19 @@ pub enum PreparationCommandKind { } #[derive(Debug, thiserror::Error)] pub enum PreparationReadyError { + #[error("preparation custody intent failed")] + Custody(#[from] CustodyError), #[error("preparation inactive or context differs")] Base(#[from] PreparationBaseError), #[error("preparation command encoding failed")] Codec(#[from] CodecError), - #[error("preparation command preparation failed")] - Command(#[source] Box>), } #[derive(Clone)] enum ExactPreparation { - Claim(PreparedCommand), + Claim(OwnedCustody), Renew { - command: PreparedCommand, - session: Arc, + command: OwnedCustody, + session: Option>, }, } #[must_use] @@ -32,6 +33,7 @@ pub struct ReadyPreparation { } #[derive(Clone)] struct PreparationRequest { + authority: PreparationAuthority, client: CellClient, target: CellTarget, check: LeaseCheck, @@ -71,6 +73,41 @@ fn validate(target: &CellTarget, request: &LeaseRequest) -> Result<(), Preparati Ok(()) } impl ReadyPreparation { + /// Reconstruct the registered original after process loss. Known results + /// remain knowledge; dispatch separately queries current lease authority. + pub async fn restore( + client: CellClient, + target: CellTarget, + operation: [u8; 16], + authority: PreparationAuthority, + ) -> Result { + let command = OwnedCustody::restore(&client, &target, operation).await?; + let (request, renew) = match command.action()? { + CustodyAction::ClaimPreparation(request) => (request, false), + CustodyAction::RenewPreparation(request) => (request, true), + _ => return Err(CustodyError::Context.into()), + }; + validate(&target, &request)?; + if !authority.matches(&target) { + return Err(PreparationBaseError::Context.into()); + } + Ok(Self { + inner: Box::new(PreparationRequest { + authority, + client, + target, + check: request.check, + exact: if renew { + ExactPreparation::Renew { + command, + session: None, + } + } else { + ExactPreparation::Claim(command) + }, + }), + }) + } /// Claim may recover an expired or previous-owner attempt. Do not require a /// local live session; authoritative execution checks the exact old token. pub async fn claim( @@ -78,15 +115,23 @@ impl ReadyPreparation { target: CellTarget, request: LeaseRequest, identity: MutationIdentity, + authority: PreparationAuthority, ) -> Result { validate(&target, &request)?; + if !authority.matches(&target) { + return Err(PreparationBaseError::Context.into()); + } let check = request.check.clone(); - let command = client - .prepare_command::(&target, identity, request) - .await - .map_err(|e| PreparationReadyError::Command(Box::new(e)))?; + let command = OwnedCustody::prepare( + &client, + &target, + CustodyAction::ClaimPreparation(request), + identity, + ) + .await?; Ok(Self { inner: Box::new(PreparationRequest { + authority, client, target, check, @@ -102,29 +147,49 @@ impl ReadyPreparation { pub(super) fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { (&self.inner.client, &self.inner.target, &self.inner.check) } - pub(super) fn pending(&self) -> PublicationError { - let evidence = match &self.inner.exact { + pub(super) fn evidence(&self) -> &cellule_runtime::PendingMutation { + match &self.inner.exact { ExactPreparation::Claim(command) => command.evidence(), ExactPreparation::Renew { command, .. } => command.evidence(), - }; - PublicationError::Preparation(InvocationError::Pending(Box::new(evidence.clone()))) + } + } + pub(super) fn pending(&self) -> PublicationError { + PublicationError::Preparation(InvocationError::Pending(Box::new(self.evidence().clone()))) } pub(super) async fn dispatch(self, recover: bool, fault: u8) -> DispatchResult { let inner = *self.inner; - let (kind, result, existing) = match inner.exact { - ExactPreparation::Claim(command) => ( - PreparationCommandKind::Claim, - super::super::exact::invoke(&inner.client, command, recover, INLINE_BYTES, fault) - .await, - None, - ), - ExactPreparation::Renew { command, session } => ( - PreparationCommandKind::Renew, - super::super::exact::invoke(&inner.client, command, recover, INLINE_BYTES, fault) - .await, - Some(session), - ), + let (kind, command, existing) = match inner.exact { + ExactPreparation::Claim(command) => (PreparationCommandKind::Claim, command, None), + ExactPreparation::Renew { command, session } => { + (PreparationCommandKind::Renew, command, session) + } }; + let guard = existing.clone(); + let result = command + .invoke(&inner.client, recover, fault, move || { + if let Some(session) = guard { + session + .live_lease() + .map_err(|_| Error::Command("preparation renewal custody inactive"))?; + } + Ok(()) + }) + .await + .map_err(|source| { + if matches!(&source, CustodyError::Stopped(_)) + && let Some(session) = &existing + { + session.fence(); + } + PublicationError::Custody { + evidence: Box::new(command.evidence().clone()), + source: Box::new(source), + } + })?; + let result = super::super::custody::project(result, |reply| match reply { + CustodyReply::Preparation(reply) => Some(reply), + _ => None, + }); let committed = match result { Ok(committed) => committed, Err(error) => { @@ -155,7 +220,10 @@ impl ReadyPreparation { if lease.token.repository != inner.check.token.repository || lease.token.operation != inner.check.token.operation || lease.token.request_digest != inner.check.token.request_digest - || lease.token == inner.check.token + || match kind { + PreparationCommandKind::Claim => lease.token == inner.check.token, + PreparationCommandKind::Renew => lease.token != inner.check.token, + } { return Err(PreparationBaseError::Context); } @@ -167,6 +235,7 @@ impl ReadyPreparation { actor: inner.check.actor.clone(), }, Some(committed.receipt), + inner.authority.clone(), ) .await?; if session.lease.base != lease.base || session.lease.format != lease.format { @@ -189,6 +258,34 @@ impl ReadyPreparation { } } impl PreparationSession { + /// Restore the registered renewal while retaining this session's permanent + /// fence and conservative clock. Restoration is also valid after fencing: + /// known outcomes remain recoverable, but absence cannot restart custody. + pub async fn restore_renewal( + self: &Arc, + ) -> Result { + let command = + OwnedCustody::restore(&self.client, &self.target, self.check.token.operation).await?; + let CustodyAction::RenewPreparation(request) = command.action()? else { + return Err(CustodyError::Context.into()); + }; + validate(&self.target, &request)?; + if request.check.token != self.check.token || request.check.actor != self.check.actor { + return Err(PreparationBaseError::Context.into()); + } + Ok(ReadyPreparation { + inner: Box::new(PreparationRequest { + authority: self.authority.clone(), + client: self.client.clone(), + target: self.target.clone(), + check: self.check.clone(), + exact: ExactPreparation::Renew { + command, + session: Some(self.clone()), + }, + }), + }) + } /// Prepare an exact renewal for service dispatch; a refused admission keeps /// the same identity. An ambiguous renewal keeps the previously observed /// deadline until resolved, and never grants custody from a recorded clock. @@ -203,20 +300,23 @@ impl PreparationSession { lease_ms, }; validate(&self.target, &request)?; - let command = self - .client - .prepare_command::(&self.target, identity, request) - .await - .map_err(|e| PreparationReadyError::Command(Box::new(e)))?; + let command = OwnedCustody::prepare( + &self.client, + &self.target, + CustodyAction::RenewPreparation(request), + identity, + ) + .await?; self.live_lease()?; Ok(ReadyPreparation { inner: Box::new(PreparationRequest { + authority: self.authority.clone(), client: self.client.clone(), target: self.target.clone(), check: self.check.clone(), exact: ExactPreparation::Renew { command, - session: self.clone(), + session: Some(self.clone()), }, }), }) diff --git a/crates/canopy-server/src/packs/publication/coordinator/recovery.rs b/crates/canopy-server/src/packs/publication/coordinator/recovery.rs index ee19df86..f3d18c6a 100644 --- a/crates/canopy-server/src/packs/publication/coordinator/recovery.rs +++ b/crates/canopy-server/src/packs/publication/coordinator/recovery.rs @@ -45,10 +45,11 @@ impl ReadyBoundRecovery { store: &ArtifactStore, ) -> Self { let client = owner.capability().0.clone(); + let authority = owner.session().authority.clone(); Self { owner, intent, - ready: ReadyRootRecovery::from_verified(registered, client, store.clone()), + ready: ReadyRootRecovery::from_verified(registered, client, store.clone(), authority), refusal, } } diff --git a/crates/canopy-server/src/packs/publication/coordinator/serving_drain.rs b/crates/canopy-server/src/packs/publication/coordinator/serving_drain.rs new file mode 100644 index 00000000..3106ee0b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator/serving_drain.rs @@ -0,0 +1,136 @@ +//! An eviction owner excludes new work while releasing its exact read pins. +use super::*; + +pub(super) struct Gate { + owner: Arc<()>, + readers: Box<[ServingToken]>, + remaining: Vec, +} + +/// This controls scheduling only. Every release still needs its private drained +/// capability, current Admin/owner and exact receiver checks. +#[must_use] +pub struct ServingDrainAdmission { + inner: Arc, + owner: Arc<()>, +} +impl Drop for ServingDrainAdmission { + fn drop(&mut self) { + let mut gate = self + .inner + .serving_drain + .lock() + .expect("serving drain admission"); + if gate + .as_ref() + .is_some_and(|gate| Arc::ptr_eq(&gate.owner, &self.owner)) + { + gate.take(); + } + drop(gate); + self.inner.drained.notify_waiters(); + // Never change State.closed or abandon any accepted command here. + } +} +impl Inner { + pub(super) fn drain_allows(&self, ready: &ReadyPublication) -> bool { + self.serving_drain + .lock() + .expect("serving drain admission") + .as_ref() + .is_none_or(|gate| { + ready + .serving_release_token() + .is_some_and(|token| gate.readers.contains(&token)) + }) + } +} +impl PublicationCoordinator { + /// Pause serving producers/borrows first and keep them paused until this + /// guard is dropped or the coordinator is closed. Busy admission is refused + /// without changing any existing command or closing the queue. + pub async fn reserve_serving_drain( + &self, + readers: &[ServingToken], + ) -> Result, PublicationScheduleError> { + if readers.len() > 16 + || readers.iter().any(|token| token.validate().is_err()) + || readers + .iter() + .enumerate() + .any(|(i, id)| readers[..i].contains(id)) + { + return Err(PublicationScheduleError::InvalidLimits); + } + for token in readers { + if crate::repository_target( + self.inner.target.tenant(), + self.inner.target.application(), + token.repository, + ) + .map_err(|_| PublicationScheduleError::Foreign)? + != self.inner.target + { + return Err(PublicationScheduleError::Foreign); + } + } + let state = self.inner.state.lock().await; + let mut gate = self + .inner + .serving_drain + .lock() + .expect("serving drain admission"); + if state.closed || state.worker || !state.jobs.is_empty() || gate.is_some() { + return Ok(None); + } + let owner = Arc::new(()); + *gate = Some(Gate { + owner: owner.clone(), + readers: readers.into(), + remaining: readers.to_vec(), + }); + Ok(Some(ServingDrainAdmission { + inner: self.inner.clone(), + owner, + })) + } +} + +impl Inner { + pub(super) fn observe_serving_release(&self, token: ServingToken) { + if let Some(gate) = self + .serving_drain + .lock() + .expect("serving drain admission") + .as_mut() + { + gate.remaining.retain(|pending| *pending != token); + } + } +} +impl ServingDrainAdmission { + /// Close only after every selected exact root has an observed successful + /// release and all admitted work has finished. A denial or uncertainty is + /// never a completed release. Failure leaves the guard and queue unchanged. + pub async fn close_if_drained(&self) -> bool { + let mut state = self.inner.state.lock().await; + let gate = self + .inner + .serving_drain + .lock() + .expect("serving drain admission"); + if !gate + .as_ref() + .is_some_and(|gate| Arc::ptr_eq(&gate.owner, &self.owner) && gate.remaining.is_empty()) + || state.worker + || !state.jobs.is_empty() + { + return false; + } + state.closed = true; + drop(gate); + drop(state); + self.inner.drained.notify_waiters(); + true + } +} diff --git a/crates/canopy-server/src/packs/publication/coordinator/work.rs b/crates/canopy-server/src/packs/publication/coordinator/work.rs index 5db60bfe..7bc2b5fd 100644 --- a/crates/canopy-server/src/packs/publication/coordinator/work.rs +++ b/crates/canopy-server/src/packs/publication/coordinator/work.rs @@ -68,14 +68,32 @@ impl PreparedCompaction { /// Immutable root completions and policy pages require registered recovery. #[must_use] pub enum ReadyPublication { + ServingRelease(ReadyServingRelease), + ServingCommand(Box), Push(ReadyCatalogPush), RootRecovery(ReadyRootRecovery), TerminalRelease(Box), + CustodyStop(Box), BoundRecovery(ReadyBoundRecovery), Compaction(ReadyCatalogCompaction), Inputs(ReadyNativeInputs), Preparation(ReadyPreparation), } +impl From for ReadyPublication { + fn from(ready: ReadyServingCommand) -> Self { + Self::ServingCommand(Box::new(ready)) + } +} +impl From for ReadyPublication { + fn from(ready: ReadyServingRelease) -> Self { + Self::ServingRelease(ready) + } +} +impl From for ReadyPublication { + fn from(ready: ReadyCustodyStop) -> Self { + Self::CustodyStop(Box::new(ready)) + } +} impl From for ReadyPublication { fn from(ready: ReadyTerminalRelease) -> Self { Self::TerminalRelease(Box::new(ready)) @@ -112,6 +130,31 @@ impl From for ReadyPublication { } } impl ReadyPublication { + pub(super) fn serving_release_token(&self) -> Option { + match self { + Self::ServingRelease(ready) => Some(ready.token()), + _ => None, + } + } + + pub(super) fn job_kind(&self) -> JobKind { + match self { + Self::CustodyStop(ready) if ready.purpose() == CustodyPurpose::Serving => { + JobKind::ServingStop + } + Self::CustodyStop(_) => JobKind::CustodyStop, + Self::ServingCommand(_) => JobKind::ServingCommand, + Self::ServingRelease(_) => JobKind::ServingRelease, + _ => JobKind::Publication, + } + } + pub(super) fn custody_original(&self) -> Option<&cellule_runtime::PendingMutation> { + match self { + Self::Preparation(ready) => Some(ready.evidence()), + Self::ServingCommand(ready) => Some(ready.evidence()), + _ => None, + } + } pub(in crate::packs::publication) fn is_policy_page(&self) -> bool { matches!(self, Self::BoundRecovery(ready) if ready.ready.is_policy_page()) } @@ -128,7 +171,10 @@ impl ReadyPublication { Self::Inputs(_) | Self::Preparation(_) | Self::RootRecovery(_) - | Self::TerminalRelease(_) => return false, + | Self::TerminalRelease(_) + | Self::CustodyStop(_) + | Self::ServingRelease(_) + | Self::ServingCommand(_) => return false, }; source.target == session.target && source.check == session.check @@ -138,6 +184,7 @@ impl ReadyPublication { } pub(super) fn reservation(&self) -> u64 { match self { + Self::ServingCommand(ready) => ready.reservation(), Self::Inputs(_) => inputs::INPUT_RESERVATION, Self::Preparation(_) => preparation::RESERVATION, Self::RootRecovery(ready) => ready.reservation(), @@ -147,9 +194,12 @@ impl ReadyPublication { } pub(super) fn dispatch_copy(&self) -> Self { match self { + Self::ServingRelease(ready) => Self::ServingRelease(ready.dispatch_copy()), + Self::ServingCommand(ready) => Self::ServingCommand(Box::new(ready.dispatch_copy())), Self::Preparation(ready) => Self::Preparation(ready.dispatch_copy()), Self::RootRecovery(ready) => Self::RootRecovery(ready.clone()), Self::TerminalRelease(ready) => Self::TerminalRelease(ready.clone()), + Self::CustodyStop(ready) => Self::CustodyStop(ready.clone()), Self::BoundRecovery(ready) => Self::BoundRecovery(ready.clone()), Self::Push(ready) => Self::Push(ReadyCatalogPush { owner: ready.owner.clone(), @@ -169,15 +219,22 @@ impl ReadyPublication { pub(super) fn class(&self) -> PublicationClass { match self { Self::Push(_) + | Self::ServingCommand(_) | Self::RootRecovery(_) | Self::BoundRecovery(_) | Self::Inputs(_) | Self::Preparation(_) => PublicationClass::Foreground, - Self::Compaction(_) | Self::TerminalRelease(_) => PublicationClass::Maintenance, + Self::Compaction(_) + | Self::TerminalRelease(_) + | Self::CustodyStop(_) + | Self::ServingRelease(_) => PublicationClass::Maintenance, } } - pub(super) fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { - match self { + pub(super) fn context(&self) -> (&CellClient, &CellTarget, BeginRequest) { + let (client, target, check) = match self { + Self::ServingRelease(ready) => return ready.context(), + Self::ServingCommand(ready) => return ready.context(), + Self::CustodyStop(ready) => return ready.context(), Self::Push(ready) => ready.owner.capability(), Self::RootRecovery(ready) => ready.capability(), Self::TerminalRelease(ready) => ready.capability(), @@ -185,13 +242,27 @@ impl ReadyPublication { Self::Inputs(ready) => ready.session.capability(), Self::Preparation(ready) => ready.capability(), Self::Compaction(ready) => ready.prepared.preparation_base().capability(), - } + }; + ( + client, + target, + BeginRequest { + repository: check.token.repository, + operation: check.token.operation, + request_digest: check.token.request_digest, + actor: check.actor.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) } pub(super) fn pending(&self) -> PublicationError { match self { + Self::ServingRelease(ready) => ready.pending(), + Self::ServingCommand(ready) => ready.pending(), Self::Preparation(ready) => ready.pending(), Self::RootRecovery(ready) => ready.pending(), Self::TerminalRelease(ready) => ready.pending(), + Self::CustodyStop(ready) => ready.pending(), Self::BoundRecovery(ready) => ready.ready.pending(), Self::Push(ready) => PublicationError::Push(InvocationError::Pending(Box::new( ready.command.evidence().clone(), @@ -205,12 +276,22 @@ impl ReadyPublication { } } pub(super) async fn dispatch(self, recover: bool, fault: u8) -> DispatchResult { - let client = self.capability().0.clone(); + let client = self.context().0.clone(); match self { + Self::ServingCommand(ready) => ready + .dispatch(recover, fault) + .await + .map(PublicationOutcome::ServingCommand), + Self::ServingRelease(ready) => ready + .dispatch(recover, fault) + .await + .map(PublicationOutcome::ServingRelease) + .map_err(PublicationError::ServingRelease), Self::Inputs(ready) => ready.dispatch(recover, fault).await, Self::Preparation(ready) => ready.dispatch(recover, fault).await, Self::RootRecovery(ready) => ready.dispatch(fault).await, Self::TerminalRelease(ready) => ready.dispatch(recover, fault).await, + Self::CustodyStop(ready) => ready.dispatch(recover, fault).await, Self::BoundRecovery(ready) => ready.dispatch(fault).await, Self::Push(ready) => super::super::exact::invoke_guarded( &client, @@ -254,6 +335,9 @@ impl ReadyPublication { #[derive(Clone, Debug)] pub enum PublicationOutcome { + ServingRelease(Committed), + ServingCommand(Committed), + Initialization(Committed), Push(Committed), RootPush(Committed), /// Original page result/receipt only; fresh guard checks remain mandatory. @@ -261,10 +345,24 @@ pub enum PublicationOutcome { Compaction(Committed), Inputs(RegisteredNativeInputs), TerminalRelease(Committed), + CustodyStop(Box), Preparation(PreparationCommandOutcome), } #[derive(Debug, thiserror::Error)] pub enum PublicationError { + #[error("serving pin release: {0}")] + ServingRelease(#[source] InvocationError), + #[error("serving custody command: {0}")] + ServingCommand(#[source] InvocationError), + #[error("publication custody intent failed")] + Custody { + evidence: Box, + source: Box, + }, + #[error("repository initialization publication: {0}")] + Initialization(#[source] InvocationError), + #[error("custody retirement: {0}")] + CustodyStop(#[source] InvocationError), #[error("terminal recovery release: {0}")] TerminalRelease(#[source] InvocationError), #[error("durable publication phase could not be observed: {source}")] @@ -297,7 +395,12 @@ impl PublicationError { } } match self { + Self::Custody { source, .. } if source.uncertain() => "pending", + Self::Custody { .. } => "not_started", Self::Recovery { .. } => "pending", + Self::ServingRelease(error) => kind(error), + Self::ServingCommand(error) => kind(error), + Self::Initialization(error) => kind(error), Self::Push(error) => kind(error), Self::RootPush(error) => kind(error), Self::PolicyPage(error) => kind(error), @@ -305,6 +408,7 @@ impl PublicationError { Self::Inputs(error) => kind(error), Self::Compaction(error) => kind(error), Self::TerminalRelease(error) => kind(error), + Self::CustodyStop(error) => kind(error), } } pub(super) fn uncertain(&self) -> bool { @@ -315,7 +419,11 @@ impl PublicationError { ) } match self { + Self::Custody { source, .. } => source.uncertain(), Self::Recovery { .. } => true, + Self::ServingRelease(error) => unknown(error), + Self::ServingCommand(error) => unknown(error), + Self::Initialization(error) => unknown(error), Self::Push(error) => unknown(error), Self::RootPush(error) => unknown(error), Self::PolicyPage(error) => unknown(error), @@ -323,6 +431,7 @@ impl PublicationError { Self::Inputs(error) => unknown(error), Self::Compaction(error) => unknown(error), Self::TerminalRelease(error) => unknown(error), + Self::CustodyStop(error) => unknown(error), } } } diff --git a/crates/canopy-server/src/packs/publication/custody/codec.rs b/crates/canopy-server/src/packs/publication/custody/codec.rs new file mode 100644 index 00000000..2f31729d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/custody/codec.rs @@ -0,0 +1,341 @@ +use super::*; +use crate::packs::directory::index::codec::fixed as wire_fixed; + +impl WireValue for CustodyAction { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::BeginPreparation(r) => { + e.write_u8(0)?; + r.encode(e) + } + Self::ClaimPreparation(r) => { + e.write_u8(1)?; + r.encode(e) + } + Self::RenewPreparation(r) => { + e.write_u8(2)?; + r.encode(e) + } + Self::BeginStaging(r) => { + e.write_u8(3)?; + r.encode(e) + } + Self::ClaimStaging(r) => { + e.write_u8(4)?; + r.encode(e) + } + Self::RenewStaging(r) => { + e.write_u8(5)?; + r.encode(e) + } + Self::BindStaging(r) => { + e.write_u8(6)?; + r.encode(e) + } + Self::AcquireServing(r) => { + e.write_u8(7)?; + r.encode(e) + } + Self::RenewServing { + request, + request_digest, + } => { + if request.check.actor.is_none() { + return Err(CodecError::Invalid( + "serving custody needs an account owner", + )); + } + e.write_u8(8)?; + request.encode(e)?; + e.write_bytes(request_digest) + } + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + Ok(match d.read_u8()? { + 0 => Self::BeginPreparation(BeginRequest::decode(d)?), + 1 => Self::ClaimPreparation(LeaseRequest::decode(d)?), + 2 => Self::RenewPreparation(LeaseRequest::decode(d)?), + 3 => Self::BeginStaging(BeginRequest::decode(d)?), + 4 => Self::ClaimStaging(LeaseRequest::decode(d)?), + 5 => Self::RenewStaging(LeaseRequest::decode(d)?), + 6 => Self::BindStaging(LeaseCheck::decode(d)?), + 7 => Self::AcquireServing(BeginRequest::decode(d)?), + 8 => { + let request = RenewServingRequest::decode(d)?; + if request.check.actor.is_none() { + return Err(CodecError::Invalid( + "serving custody needs an account owner", + )); + } + Self::RenewServing { + request, + request_digest: wire_fixed(d)?, + } + } + _ => return Err(CodecError::Invalid("custody action purpose")), + }) + } +} +impl WireValue for CustodyRequest { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.step > MAX_STEPS || (self.step == 0) != self.previous.is_none() { + return Err(CodecError::Invalid("custody predecessor step")); + } + e.write_bytes(DOMAIN)?; + e.write_u32(self.step)?; + e.write_bool(self.previous.is_some())?; + if let Some(previous) = self.previous { + e.write_bytes(&previous)?; + } + self.action.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("custody request purpose")); + } + let value = Self { + step: d.read_u32()?, + previous: if d.read_bool()? { + Some(wire_fixed(d)?) + } else { + None + }, + action: CustodyAction::decode(d)?, + }; + value.encode(&mut BoundedEncoder::new(INPUT_BYTES)?)?; + Ok(value) + } +} +impl WireValue for CustodyReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Preparation(reply) => { + e.write_u8(0)?; + reply.encode(e) + } + Self::Staging(reply) => { + e.write_u8(1)?; + reply.encode(e) + } + Self::Serving(reply) => { + e.write_u8(2)?; + reply.encode(e) + } + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 0 => Ok(Self::Preparation(PreparationReply::decode(d)?)), + 1 => Ok(Self::Staging(StagingReply::decode(d)?)), + 2 => Ok(Self::Serving(ServingReply::decode(d)?)), + _ => Err(CodecError::Invalid("custody reply purpose")), + } + } +} +impl WireValue for Header { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.operation == [0; 16] + || self.step > MAX_STEPS + || (self.step == 0) != self.previous.is_none() + || validate_component(&self.actor).is_err() + { + return Err(CodecError::Invalid("custody header identity")); + } + e.write_bytes(DOMAIN)?; + e.write_u8(self.purpose.number())?; + e.write_bytes(&self.tenant)?; + e.write_bytes(&self.application)?; + e.write_bytes(self.incarnation.as_bytes())?; + self.stamp.encode(e)?; + e.write_bytes(&self.repository)?; + e.write_bytes(&self.operation)?; + e.write_bytes(&self.request_digest)?; + e.write_text(&self.actor)?; + e.write_u32(self.step)?; + e.write_bool(self.previous.is_some())?; + if let Some(previous) = self.previous { + e.write_bytes(&previous)?; + } + e.write_bytes(&self.bundle_digest) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("custody header purpose")); + } + let value = Self { + purpose: CustodyPurpose::parse(d.read_u8()?)?, + tenant: wire_fixed(d)?, + application: wire_fixed(d)?, + incarnation: IncarnationId::from_bytes(wire_fixed(d)?), + stamp: Stamp::decode(d)?, + repository: wire_fixed(d)?, + operation: wire_fixed(d)?, + request_digest: wire_fixed(d)?, + actor: d.read_text()?.into(), + step: d.read_u32()?, + previous: if d.read_bool()? { + Some(wire_fixed(d)?) + } else { + None + }, + bundle_digest: wire_fixed(d)?, + }; + value.encode(&mut BoundedEncoder::new(CERTIFICATE_BYTES)?)?; + Ok(value) + } +} +impl WireValue for CustodyIntent { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.header()?; + self.request()?; + self.certificate.encode(e)?; + e.write_bytes(&self.snapshot.to_bytes()?)?; + e.write_bytes(&self.body) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + certificate: CertificateEnvelope::decode(d)?, + snapshot: PreparedCommandSnapshot::from_bytes(d.read_bytes()?)?, + body: d.read_bytes()?.to_vec(), + }; + value.encode(&mut BoundedEncoder::new(INTENT_BYTES)?)?; + Ok(value) + } +} + +pub(super) fn validate_phase(phase: &Recorded, request: &CustodyRequest) -> Result<(), CodecError> { + let reply: CustodyReply = phase.decode_reply()?; + if phase.rejected() != reply.rejected() + || request.action.staging() != matches!(reply, CustodyReply::Staging(_)) + || (request.action.purpose() == CustodyPurpose::Serving) + != matches!(reply, CustodyReply::Serving(_)) + { + return Err(CodecError::Invalid("custody result purpose differs")); + } + let token = match reply { + CustodyReply::Preparation(PreparationReply::Granted(lease)) => { + lease.base.validate()?; + Some(lease.token) + } + CustodyReply::Staging(StagingReply::Granted(lease)) => Some(lease.token), + CustodyReply::Serving(ServingReply::Granted(lease)) => { + lease.validate()?; + let (repository, reader, _, _) = request.action.identity(); + if lease.token.repository != repository + || lease.token.reader != reader + || lease.token.admission_sequence > phase.sequence() + { + return Err(CodecError::Invalid("serving custody grant binding differs")); + } + match &request.action { + CustodyAction::AcquireServing(_) + if lease.token.admission_sequence == phase.sequence() => {} + CustodyAction::RenewServing { request, .. } + if request.check.token == lease.token => {} + _ => return Err(CodecError::Invalid("serving custody original differs")), + } + None + } + _ => None, + }; + if let Some(token) = token { + let (repository, operation, digest, _) = request.action.identity(); + if token.repository != repository + || token.operation != operation + || token.request_digest != digest + || token.attempt > phase.sequence() + { + return Err(CodecError::Invalid("custody grant binding differs")); + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn serving_wire_requires_account_exact_framing_and_v2_purpose() -> Result<(), CodecError> { + let begin = BeginRequest { + repository: *uuid::Uuid::new_v4().as_bytes(), + operation: [2; 16], + request_digest: [3; 32], + actor: "viewer".into(), + lease_ms: DEFAULT_LEASE_MS, + }; + let token = ServingToken { + repository: begin.repository, + reader: begin.operation, + owner: OwnerFence { + incarnation: IncarnationId::from_bytes([4; 16]), + epoch: 1, + }, + admission_sequence: 5, + generation: 1, + }; + let renew = RenewServingRequest { + check: ServingCheck { + token, + actor: Some("viewer".into()), + }, + lease_ms: DEFAULT_LEASE_MS, + }; + for action in [ + CustodyAction::AcquireServing(begin.clone()), + CustodyAction::RenewServing { + request: renew.clone(), + request_digest: begin.request_digest, + }, + ] { + let request = CustodyRequest { + step: 0, + previous: None, + action, + }; + let bytes = encode(&request, INPUT_BYTES)?; + assert_eq!(decode::(&bytes, INPUT_BYTES)?, request); + for length in 0..bytes.len() { + assert!(decode::(&bytes[..length], INPUT_BYTES).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(decode::(&trailing, INPUT_BYTES).is_err()); + let mut old = bytes; + let version = old + .windows(3) + .position(|part| part == b"v2\0") + .expect("domain version"); + old[version + 1] = b'1'; + assert!(decode::(&old, INPUT_BYTES).is_err()); + for reply in [ + CustodyReply::Preparation(PreparationReply::Denied( + PreparationDenial::Unauthorized, + )), + CustodyReply::Staging(StagingReply::Denied(PreparationDenial::Unauthorized)), + ] { + let phase = Recorded::new(6, true, encode(&reply, 512)?)?; + assert!(validate_phase(&phase, &request).is_err()); + } + let reply = CustodyReply::Serving(ServingReply::Denied(ServingDenial::Unauthorized)); + let bytes = encode(&reply, 512)?; + assert!(validate_phase(&Recorded::new(6, false, bytes.clone())?, &request).is_err()); + validate_phase(&Recorded::new(6, true, bytes)?, &request)?; + } + let mut anonymous = renew; + anonymous.check.actor = None; + assert!( + encode( + &CustodyAction::RenewServing { + request: anonymous, + request_digest: [3; 32] + }, + INPUT_BYTES + ) + .is_err() + ); + assert!(CustodyPurpose::parse(2).is_err()); + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/publication/custody/commands.rs b/crates/canopy-server/src/packs/publication/custody/commands.rs new file mode 100644 index 00000000..33afda22 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/custody/commands.rs @@ -0,0 +1,191 @@ +//! First-writer registration and domain result commit in the Repository Cell. +use super::*; + +pub struct RegisterCustodyIntent; +impl Command for RegisterCustodyIntent { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 41; + const CODEC_VERSION: u32 = 2; + type Input = CustodyIntent; + type Output = RootRecoveryReply; + fn execute( + context: &mut CommandContext<'_, '_>, + intent: Self::Input, + ) -> cellule_runtime::Result> { + let deny = |reason| Ok(CommandResult::Rejected(RootRecoveryReply::Denied(reason))); + let seed = super::super::attestation::seed(&context.sql(&SqlBatch { + statements: vec![seed_statement()], + })?)?; + let header = intent.validate(context.target(), &seed)?; + let request = intent.request()?; + let bytes = intent.encoded()?; + let previous = context.sql(&SqlBatch { + statements: vec![row_statement(header.key(), None), seed_statement()], + })?; + let previous = from_sets(&previous, context.target(), header.key())?; + if let Some(previous) = &previous { + if previous.intent == intent { + // Exact knowledge is idempotent even after permission/owner loss. + return Ok(CommandResult::Success(RootRecoveryReply::Registered)); + } + let old = previous.intent.header()?; + if old.step >= header.step + || !previous.closed() + || old.step.checked_add(1) != Some(header.step) + || header.previous != Some(*blake3::hash(&previous.intent.encoded()?).as_bytes()) + || old.actor != header.actor + || old.request_digest != header.request_digest + || old.repository != header.repository + { + return deny(PreparationDenial::Conflict); + } + } else if header.step != 0 || !request.action.begin() { + return deny(PreparationDenial::Missing); + } + if header.incarnation != context.owner_fence().incarnation { + return deny(PreparationDenial::Stale); + } + if intent.snapshot.evidence().identity().expires_at_ms <= now(context.now_ms())? { + return deny(PreparationDenial::Expired); + } + if super::super::commands::authorized( + context, + header.repository, + &header.actor, + if header.purpose == CustodyPurpose::Serving { + TokenScope::Read + } else { + TokenScope::Write + }, + )? + .is_none() + { + return deny(PreparationDenial::Unauthorized); + } + let pending = context.sql(&statement( + "SELECT count(*) FROM (SELECT operation FROM catalog_custody_commands WHERE phase IS NULL AND stopped IS NULL LIMIT ?1)", + vec![number(MAX_OPERATIONS)?], + ))?; + let Some([SqlValue::Integer(pending)]) = rows(&pending)?.first().map(Vec::as_slice) else { + return Err(Error::Command("custody pending count absent")); + }; + if *pending >= MAX_OPERATIONS as i64 { + return deny(PreparationDenial::Capacity); + } + super::super::publish::changed(context.sql(&statement( + "INSERT INTO catalog_custody_commands(operation,step,incarnation,request_id,intent,phase,purpose) VALUES(?1,?2,?3,?4,?5,NULL,?6)", + vec![blob(header.operation),number(u64::from(header.step))?,blob(header.incarnation.as_bytes()), + blob(intent.snapshot.evidence().identity().request_id.as_bytes()), SqlValue::Blob(bytes), number(u64::from(header.purpose.number()))?], + ))?)?; + Ok(CommandResult::Success(RootRecoveryReply::Registered)) + } +} + +/// The only receiver of the new custody protocol. Domain methods reuse the +/// existing allocator, pins, authorization and phase-specific validation. +pub struct ExecuteCustody; +impl Command for ExecuteCustody { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 42; + const CODEC_VERSION: u32 = 2; + type Input = CustodyRequest; + type Output = CustodyReply; + fn execute( + context: &mut CommandContext<'_, '_>, + request: Self::Input, + ) -> cellule_runtime::Result> { + let (_, operation, _, _) = request.action.identity(); + let sets = context.sql(&SqlBatch { + statements: vec![ + row_statement(request.action.key(), Some(request.step)), + seed_statement(), + ], + })?; + let saved = from_sets(&sets, context.target(), request.action.key())? + .ok_or(Error::Command("custody command is not registered"))?; + let header = saved.intent.header()?; + let evidence = context + .mutation_evidence() + .ok_or(Error::Command("custody command lacks mutation evidence"))?; + if header.stamp != Stamp::of(&evidence) + || header.incarnation != evidence.incarnation() + || saved.intent.request()? != request + || saved.closed() + { + return Err(Error::Command( + "custody command differs from its original intent", + )); + } + let output = match request.action.clone() { + CustodyAction::BeginPreparation(input) => { + prep(BeginPreparation::execute(context, input)?) + } + CustodyAction::ClaimPreparation(input) => { + prep(ClaimPreparation::execute(context, input)?) + } + CustodyAction::RenewPreparation(input) => { + prep(RenewPreparation::execute(context, input)?) + } + CustodyAction::BeginStaging(input) => stage(BeginStaging::execute(context, input)?), + CustodyAction::ClaimStaging(input) => stage(ClaimStaging::execute(context, input)?), + CustodyAction::RenewStaging(input) => stage(RenewStaging::execute(context, input)?), + CustodyAction::BindStaging(input) => prep(BindStaging::execute(context, input)?), + CustodyAction::AcquireServing(input) => serving(AcquireServingPin::execute( + context, + AcquireServingRequest { + repository: input.repository, + reader: input.operation, + actor: Some(input.actor), + lease_ms: input.lease_ms, + }, + )?), + CustodyAction::RenewServing { request, .. } => { + serving(RenewServingPin::execute(context, request)?) + } + }; + let phase = Recorded::new(context.sequence(), output.rejected(), encode(&output, 512)?)?; + codec::validate_phase(&phase, &request)?; + let grant = match &output { + CustodyReply::Preparation(PreparationReply::Granted(lease)) => { + Some((lease.token.owner, lease.token.attempt)) + } + CustodyReply::Staging(StagingReply::Granted(lease)) => { + Some((lease.token.owner, lease.token.attempt)) + } + CustodyReply::Serving(ServingReply::Granted(lease)) => { + Some((lease.token.owner, lease.token.admission_sequence)) + } + _ => None, + }; + let grant_incarnation = grant.map_or(SqlValue::Null, |(owner, _)| { + blob(owner.incarnation.as_bytes()) + }); + let grant_attempt = grant + .map(|(_, attempt)| number(attempt)) + .transpose()? + .unwrap_or(SqlValue::Null); + super::super::publish::changed(context.sql(&statement( + "UPDATE catalog_custody_commands SET phase=?1,granted_incarnation=?5,granted_attempt=?6 WHERE operation=?2 AND step=?3 AND intent=?4 AND phase IS NULL AND stopped IS NULL AND purpose=?7", + vec![SqlValue::Blob(encode(&phase, 1024)?),blob(operation),number(u64::from(request.step))?,SqlValue::Blob(saved.intent.encoded()?),grant_incarnation,grant_attempt,number(u64::from(header.purpose.number()))?], + ))?)?; + // Trusted denials commit the original phase alongside SDK acceptance; + // the private service normalizes them back to Rejected at its boundary. + Ok(CommandResult::Success(output)) + } +} +fn prep(result: CommandResult) -> CustodyReply { + CustodyReply::Preparation(match result { + CommandResult::Success(r) | CommandResult::Rejected(r) => r, + }) +} +fn stage(result: CommandResult) -> CustodyReply { + CustodyReply::Staging(match result { + CommandResult::Success(r) | CommandResult::Rejected(r) => r, + }) +} + +fn serving(result: CommandResult) -> CustodyReply { + CustodyReply::Serving(match result { + CommandResult::Success(r) | CommandResult::Rejected(r) => r, + }) +} diff --git a/crates/canopy-server/src/packs/publication/custody/dispatch.rs b/crates/canopy-server/src/packs/publication/custody/dispatch.rs new file mode 100644 index 00000000..d125aaf0 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/custody/dispatch.rs @@ -0,0 +1,233 @@ +//! Service-owned original registration and custody command, never a fresh retry. +use super::*; +use cellule_runtime::PreparedCommand; + +// Retained and dispatch owners each hold intent + registrar body (four copies). +// Registrar transport and bounded query decode add two more intent ceilings. +// Original execution bodies and phase/reply decoding have independent ceilings. +pub(in crate::packs::publication) const RESERVATION: u64 = + 6 * INTENT_BYTES as u64 + 2 * INPUT_BYTES as u64 + 2 * 1024; + +#[derive(Clone)] +pub(in crate::packs::publication) struct OwnedCustody { + prepared: PreparedCustody, + registration: Option>, +} +/// A read probe retains no original command body or registrar. Its exact +/// fingerprint prevents a later logical successor from hiding this ordinal. +#[derive(Clone)] +pub(in crate::packs::publication) struct CustodyStopProbe { + target: CellTarget, + operation: [u8; 16], + purpose: CustodyPurpose, + step: u32, + digest: [u8; 32], + evidence: PendingMutation, +} +impl CustodyStopProbe { + pub(in crate::packs::publication) fn evidence(&self) -> &PendingMutation { + &self.evidence + } + pub(in crate::packs::publication) async fn observed( + &self, + client: &CellClient, + ) -> Result { + let Some(saved) = load( + client, + &self.target, + CustodyKey { + purpose: self.purpose, + operation: self.operation, + }, + Some(self.step), + ) + .await? + else { + return Ok(false); + }; + if *blake3::hash(&saved.intent.encoded()?).as_bytes() != self.digest + || saved.evidence() != &self.evidence + { + return Err(CustodyError::Context); + } + Ok(saved.stop_fact().is_some()) + } +} +impl OwnedCustody { + /// Observe only this exact accepted ordinal. This must never execute an + /// absent original or substitute the latest renewal's receipt. + pub(in crate::packs::publication) async fn serving_grant( + &self, + client: &CellClient, + ) -> Result { + let header = self.prepared.intent.header()?; + if !matches!(self.action()?, CustodyAction::AcquireServing(_)) { + return Err(CustodyError::Context); + } + let saved = load( + client, + self.evidence().target(), + header.key(), + Some(header.step), + ) + .await? + .ok_or(CustodyError::Context)?; + if saved.intent != self.prepared.intent || saved.stopped.is_some() { + return Err(CustodyError::Context); + } + let phase = saved.phase.ok_or(CustodyError::Context)?; + let committed = phase.committed::(self.evidence())?; + match committed.output { + CustodyReply::Serving(ServingReply::Granted(lease)) + if !phase.rejected() + && lease.token.repository == header.repository + && lease.token.reader == header.operation + && lease.token.admission_sequence == committed.receipt.commit_sequence => + { + Ok(*lease) + } + _ => Err(CustodyError::Context), + } + } + pub(in crate::packs::publication) fn stop_probe( + &self, + ) -> Result { + let header = self.prepared.intent.header()?; + Ok(CustodyStopProbe { + target: self.evidence().target().clone(), + operation: header.operation, + purpose: header.purpose, + step: header.step, + digest: *blake3::hash(&self.prepared.intent.encoded()?).as_bytes(), + evidence: self.evidence().clone(), + }) + } + pub(in crate::packs::publication) async fn prepare( + client: &CellClient, + target: &CellTarget, + action: CustodyAction, + identity: MutationIdentity, + ) -> Result { + let prepared = PreparedCustody::prepare(client, target, action, identity).await?; + let registration = client + .prepare_command::( + target, + crate::server::mutation_identity() + .map_err(|error| CustodyError::Clock(Box::new(error)))?, + prepared.intent.clone(), + ) + .await + .map_err(|error| CustodyError::Registration(Box::new(error)))?; + Ok(Self { + prepared, + registration: Some(registration), + }) + } + pub(in crate::packs::publication) async fn restore( + client: &CellClient, + target: &CellTarget, + operation: [u8; 16], + ) -> Result { + Self::restore_for(client, target, CustodyPurpose::Creating, operation).await + } + pub(in crate::packs::publication) async fn restore_for( + client: &CellClient, + target: &CellTarget, + purpose: CustodyPurpose, + operation: [u8; 16], + ) -> Result { + let registered = load(client, target, CustodyKey { purpose, operation }, None) + .await? + .ok_or(CustodyError::Context)?; + Ok(Self { + prepared: PreparedCustody { + intent: registered.intent, + }, + registration: None, + }) + } + pub(in crate::packs::publication) fn action(&self) -> Result { + Ok(self.prepared.intent.request()?.action) + } + pub(in crate::packs::publication) fn evidence(&self) -> &PendingMutation { + self.prepared.evidence() + } + #[cfg(test)] + pub(in crate::packs::publication) fn registration_evidence(&self) -> Option<&PendingMutation> { + self.registration.as_ref().map(PreparedCommand::evidence) + } + async fn persist( + &self, + client: &CellClient, + recover: bool, + fault: u8, + ) -> Result { + let header = self.prepared.intent.header()?; + let target = self.evidence().target(); + if let Some(saved) = load(client, target, header.key(), Some(header.step)).await? { + return if saved.intent == self.prepared.intent { + if let Some(fact) = saved.stop_fact() { + return Err(CustodyError::Stopped(Box::new(fact))); + } + Ok(saved) + } else { + Err(CustodyError::Context) + }; + } + let registration = self.registration.as_ref().ok_or(CustodyError::Context)?; + let registration_fault = match fault { + 4 => 1, + 5 => 2, + 6 => 3, + _ => 0, + }; + let result = super::super::exact::invoke( + client, + registration.clone(), + recover, + 4096, + registration_fault, + ) + .await; + // Injected lost registrar replies remain uncertain until the observer + // requests recovery; ordinary network loss can use a durable pointer. + if registration_fault != 0 { + result.map_err(|error| CustodyError::Registration(Box::new(error)))?; + } else if let Some(saved) = load(client, target, header.key(), Some(header.step)).await? { + return if saved.intent == self.prepared.intent { + Ok(saved) + } else { + Err(CustodyError::Context) + }; + } else { + result.map_err(|error| CustodyError::Registration(Box::new(error)))?; + } + Err(CustodyError::Context) + } + pub(in crate::packs::publication) async fn invoke( + &self, + client: &CellClient, + recover: bool, + fault: u8, + before_execute: impl FnOnce() -> Result<(), Error> + Send, + ) -> Result, InvocationError>, CustodyError> { + let saved = self.persist(client, recover, fault).await?; + let execution_fault = if fault <= 3 { fault } else { 0 }; + if execution_fault == 1 { + return Ok(Err(InvocationError::Pending(Box::new( + self.evidence().clone(), + )))); + } + let result = saved.recover_guarded(client, before_execute).await; + if execution_fault == 2 { + return Ok(Err(InvocationError::Pending(Box::new( + self.evidence().clone(), + )))); + } + assert_ne!( + execution_fault, 3, + "injected custody command panic after execution" + ); + Ok(result) + } +} diff --git a/crates/canopy-server/src/packs/publication/custody/mod.rs b/crates/canopy-server/src/packs/publication/custody/mod.rs new file mode 100644 index 00000000..b56f92e0 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/custody/mod.rs @@ -0,0 +1,718 @@ +//! Exact custody commands, including the interval before an artifact namespace exists. +//! This is command metadata, never an object inventory or a custody permission. +use super::{ + certificate::CertificateEnvelope, + recovery::{Stamp, phase::Recorded}, + sql::*, + *, +}; +use cellule_runtime::{ + CellClient, CellTarget, Committed, InvocationError, MutationIdentity, PendingMutation, + PreparedCommandSnapshot, primitives::sql::SqlCell, +}; +mod codec; +mod commands; +mod dispatch; +mod scan; +mod stop; +pub use commands::{ExecuteCustody, RegisterCustodyIntent}; +pub(super) use dispatch::{OwnedCustody, RESERVATION}; +pub use scan::{CustodyScanStats, CustodySupervisor}; +pub use stop::{ + CustodyStopFact, CustodyStopInput, CustodyStopOutcome, CustodyStopReply, ReadyCustodyStop, + StopCustodyIntent, +}; + +const INPUT_BYTES: u32 = 1024; +const INTENT_BYTES: u32 = 4096; +const MAX_STEPS: u32 = 65_535; +const DOMAIN: &[u8] = b"canopy.custody-command-intent.v2\0"; + +/// Creating and serving requests retain separate exact command histories. +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub enum CustodyPurpose { + Creating, + Serving, +} +impl CustodyPurpose { + fn number(self) -> u8 { + match self { + Self::Creating => 0, + Self::Serving => 1, + } + } + fn parse(value: u8) -> Result { + match value { + 0 => Ok(Self::Creating), + 1 => Ok(Self::Serving), + _ => Err(CodecError::Invalid("custody purpose")), + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] +struct CustodyKey { + purpose: CustodyPurpose, + operation: [u8; 16], +} +impl From<[u8; 16]> for CustodyKey { + fn from(operation: [u8; 16]) -> Self { + Self { + purpose: CustodyPurpose::Creating, + operation, + } + } +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CustodyAction { + BeginPreparation(BeginRequest), + ClaimPreparation(LeaseRequest), + RenewPreparation(LeaseRequest), + BeginStaging(BeginRequest), + ClaimStaging(LeaseRequest), + RenewStaging(LeaseRequest), + BindStaging(LeaseCheck), + AcquireServing(BeginRequest), + RenewServing { + request: RenewServingRequest, + request_digest: [u8; 32], + }, +} +impl CustodyAction { + fn identity(&self) -> ([u8; 16], [u8; 16], [u8; 32], &str) { + match self { + Self::BeginPreparation(r) | Self::BeginStaging(r) | Self::AcquireServing(r) => { + (r.repository, r.operation, r.request_digest, &r.actor) + } + Self::ClaimPreparation(r) + | Self::RenewPreparation(r) + | Self::ClaimStaging(r) + | Self::RenewStaging(r) => identity_of(&r.check), + Self::BindStaging(r) => identity_of(r), + Self::RenewServing { + request, + request_digest, + } => ( + request.check.token.repository, + request.check.token.reader, + *request_digest, + request.check.actor.as_deref().unwrap_or(""), + ), + } + } + fn purpose(&self) -> CustodyPurpose { + if matches!(self, Self::AcquireServing(_) | Self::RenewServing { .. }) { + CustodyPurpose::Serving + } else { + CustodyPurpose::Creating + } + } + fn key(&self) -> CustodyKey { + CustodyKey { + purpose: self.purpose(), + operation: self.identity().1, + } + } + fn staging(&self) -> bool { + matches!( + self, + Self::BeginStaging(_) | Self::ClaimStaging(_) | Self::RenewStaging(_) + ) + } + fn begin(&self) -> bool { + matches!( + self, + Self::BeginPreparation(_) | Self::BeginStaging(_) | Self::AcquireServing(_) + ) + } +} +fn identity_of(check: &LeaseCheck) -> ([u8; 16], [u8; 16], [u8; 32], &str) { + ( + check.token.repository, + check.token.operation, + check.token.request_digest, + &check.actor, + ) +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CustodyReply { + Preparation(PreparationReply), + Staging(StagingReply), + Serving(ServingReply), +} +impl CustodyReply { + fn rejected(&self) -> bool { + matches!( + self, + Self::Preparation(PreparationReply::Denied(_)) + | Self::Staging(StagingReply::Denied(_)) + | Self::Serving(ServingReply::Denied(_)) + ) + } +} + +/// Private ordinal/predecessor fields prevent a factory from replacing an +/// unsettled head. Decoded values remain untrusted until receiver verification. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CustodyRequest { + step: u32, + previous: Option<[u8; 32]>, + action: CustodyAction, +} +#[derive(Clone, Debug, PartialEq, Eq)] +struct Header { + purpose: CustodyPurpose, + tenant: [u8; 16], + application: [u8; 16], + incarnation: IncarnationId, + stamp: Stamp, + repository: [u8; 16], + operation: [u8; 16], + request_digest: [u8; 32], + actor: String, + step: u32, + previous: Option<[u8; 32]>, + bundle_digest: [u8; 32], +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CustodyIntent { + certificate: CertificateEnvelope, + snapshot: PreparedCommandSnapshot, + body: Vec, +} +impl CustodyIntent { + fn request(&self) -> Result { + decode(&self.body, INPUT_BYTES) + } + fn header(&self) -> Result { + self.certificate.data() + } + fn validate(&self, target: &CellTarget, seed: &[u8; 32]) -> cellule_runtime::Result

{ + let header = self.header()?; + let request = self.request()?; + let evidence = self.snapshot.evidence(); + let (repository, operation, digest, actor) = request.action.identity(); + if header.purpose != request.action.purpose() + || !self.certificate.authenticated(seed) + || header.tenant != *target.tenant().as_bytes() + || header.application != *target.application().as_bytes() + || crate::repository_target(target.tenant(), target.application(), repository)? + != *target + || evidence.target() != target + || header.incarnation != evidence.incarnation() + || header.stamp != Stamp::of(evidence) + || header.repository != repository + || header.operation != operation + || header.request_digest != digest + || header.actor != actor + || header.step != request.step + || header.previous != request.previous + || header.bundle_digest != bundle_digest(&self.snapshot, &self.body)? + { + return Err(Error::Command("custody intent binding differs")); + } + Ok(header) + } + fn encoded(&self) -> Result, CodecError> { + encode(self, INTENT_BYTES) + } +} + +#[derive(Debug, thiserror::Error)] +pub enum CustodyError { + #[error("custody command clock failed")] + Clock(#[source] Box), + #[error("custody command encoding failed")] + Codec(#[from] CodecError), + #[error("custody command binding failed")] + Capability(#[from] Error), + #[error("custody command query failed")] + Query(#[source] Box>>), + #[error("custody command preparation failed")] + Preparation(#[source] Box>), + #[error("custody intent registration failed")] + Registration(#[source] Box>), + #[error("custody command has an unsettled predecessor")] + Unsettled(Box), + #[error("custody command head differs")] + Context, + #[error("custody original was retired without an execution result")] + Stopped(Box), + #[error("custody owner observation failed")] + Owner(#[source] Box), + #[error("custody stop preparation failed")] + StopPreparation(#[source] Box>), + #[error("invalid custody scan limits")] + InvalidScanLimits, +} + +impl CustodyError { + pub(super) fn uncertain(&self) -> bool { + match self { + Self::Registration(error) => matches!( + &**error, + InvocationError::Pending(_) | InvocationError::InvalidPublishedResult { .. } + ), + Self::Preparation(error) => matches!( + &**error, + InvocationError::Pending(_) | InvocationError::InvalidPublishedResult { .. } + ), + Self::Clock(_) + | Self::Stopped(_) + | Self::Owner(_) + | Self::InvalidScanLimits + | Self::StopPreparation(_) => false, + // Failure to authenticate or observe metadata is never proof of + // absence. Keep the owned original until its disposition is known. + Self::Query(_) + | Self::Codec(_) + | Self::Capability(_) + | Self::Unsettled(_) + | Self::Context => true, + } + } +} + +/// Retains the original SDK snapshot/body even if registration loses its reply. +/// Registration must become discoverable before this command can execute. +#[derive(Clone)] +#[must_use] +pub struct PreparedCustody { + intent: CustodyIntent, +} +#[derive(Clone)] +pub struct RegisteredCustody { + intent: CustodyIntent, + phase: Option, + stopped: Option, +} + +fn encode(value: &impl WireValue, limit: u32) -> Result, CodecError> { + let mut encoder = BoundedEncoder::new(limit)?; + value.encode(&mut encoder)?; + Ok(encoder.finish()) +} +fn decode(bytes: &[u8], limit: u32) -> Result { + let mut decoder = BoundedDecoder::new(bytes, limit)?; + let value = T::decode(&mut decoder)?; + decoder.finish()?; + Ok(value) +} +fn bundle_digest(snapshot: &PreparedCommandSnapshot, body: &[u8]) -> Result<[u8; 32], CodecError> { + let mut hash = blake3::Hasher::new(); + hash.update(DOMAIN); + let bytes = snapshot.to_bytes()?; + for part in [bytes.as_slice(), body] { + hash.update(&(part.len() as u64).to_be_bytes()); + hash.update(part); + } + Ok(*hash.finalize().as_bytes()) +} +fn seed_statement() -> SqlStatement { + SqlStatement { + sql: "SELECT push_cert_seed FROM repository_identity WHERE singleton=1".into(), + parameters: vec![], + } +} +impl Header { + fn key(&self) -> CustodyKey { + CustodyKey { + purpose: self.purpose, + operation: self.operation, + } + } +} +fn row_statement(key: impl Into, step: Option) -> SqlStatement { + let key = key.into(); + let mut parameters = vec![ + blob(key.operation), + SqlValue::Integer(i64::from(key.purpose.number())), + ]; + let sql = if let Some(step) = step { + parameters.push(SqlValue::Integer(i64::from(step))); + "SELECT step,incarnation,request_id,intent,phase,stopped FROM catalog_custody_commands WHERE operation=?1 AND purpose=?2 AND step=?3" + } else { + "SELECT step,incarnation,request_id,intent,phase,stopped FROM catalog_custody_commands WHERE operation=?1 AND purpose=?2 ORDER BY step DESC LIMIT 1" + }; + SqlStatement { + sql: sql.into(), + parameters, + } +} +fn from_sets( + sets: &[SqlResultSet], + target: &CellTarget, + key: impl Into, +) -> cellule_runtime::Result> { + let key = key.into(); + let Some(row) = rows(sets)?.first() else { + return Ok(None); + }; + let [ + SqlValue::Integer(step), + incarnation, + request_id, + SqlValue::Blob(bytes), + phase, + stopped, + ] = row.as_slice() + else { + return Err(Error::Command("invalid custody command row")); + }; + let intent: CustodyIntent = decode(bytes, INTENT_BYTES)?; + let seed = + super::attestation::seed(sets.get(1..).ok_or(Error::Command("custody seed absent"))?)?; + let header = intent.validate(target, &seed)?; + if i64::from(header.step) != *step + || header.key() != key + || fixed::<16>(incarnation)? != *header.incarnation.as_bytes() + || fixed::<16>(request_id)? != *intent.snapshot.evidence().identity().request_id.as_bytes() + { + return Err(Error::Command("custody command row binding differs")); + } + let phase = match phase { + SqlValue::Null => None, + SqlValue::Blob(bytes) => Some(decode::(bytes, 1024)?), + _ => return Err(Error::Command("invalid custody command phase")), + }; + if let Some(phase) = &phase { + codec::validate_phase(phase, &intent.request()?)?; + } + let stopped = stop::record(stopped, &intent, &seed)?; + if phase.is_some() && stopped.is_some() { + return Err(Error::Command("custody execution and retirement coexist")); + } + Ok(Some(RegisteredCustody { + intent, + phase, + stopped, + })) +} +async fn load( + client: &CellClient, + target: &CellTarget, + key: impl Into, + step: Option, +) -> Result, CustodyError> { + let key = key.into(); + let sql = SqlCell::::new(client.clone(), target.clone())?; + let output = sql + .query( + None, + SqlBatch { + statements: vec![row_statement(key, step), seed_statement()], + }, + ) + .await + .map_err(|error| CustodyError::Query(Box::new(error)))?; + Ok(from_sets(&output.output, target, key)?) +} +impl PreparedCustody { + pub async fn prepare( + client: &CellClient, + target: &CellTarget, + action: CustodyAction, + identity: MutationIdentity, + ) -> Result { + let (repository, _, digest, actor) = action.identity(); + if crate::repository_target(target.tenant(), target.application(), repository)? != *target { + return Err(CustodyError::Context); + } + let head = load(client, target, action.key(), None).await?; + let (step, previous) = if let Some(head) = head { + let header = head.intent.header()?; + if header.repository != repository + || header.request_digest != digest + || header.actor != actor + { + return Err(CustodyError::Context); + } + if !head.closed() { + return Err(CustodyError::Unsettled(Box::new(head.evidence().clone()))); + } + ( + header.step.checked_add(1).ok_or(CustodyError::Context)?, + Some(*blake3::hash(&head.intent.encoded()?).as_bytes()), + ) + } else { + (0, None) + }; + let request = CustodyRequest { + step, + previous, + action, + }; + let command = client + .prepare_command::(target, identity, request) + .await + .map_err(|error| CustodyError::Preparation(Box::new(error)))?; + let snapshot = command.snapshot(); + let body = command.input_bytes().to_vec(); + let request: CustodyRequest = decode(&body, INPUT_BYTES)?; + let (repository, operation, request_digest, actor) = request.action.identity(); + let header = Header { + purpose: request.action.purpose(), + tenant: *target.tenant().as_bytes(), + application: *target.application().as_bytes(), + incarnation: command.evidence().incarnation(), + stamp: Stamp::of(command.evidence()), + repository, + operation, + request_digest, + actor: actor.into(), + step, + previous, + bundle_digest: bundle_digest(&snapshot, &body)?, + }; + let sql = SqlCell::::new(client.clone(), target.clone())?; + let output = sql + .query( + None, + SqlBatch { + statements: vec![seed_statement()], + }, + ) + .await + .map_err(|error| CustodyError::Query(Box::new(error)))?; + let seed = super::attestation::seed(&output.output)?; + let intent = CustodyIntent { + certificate: CertificateEnvelope::seal(&header, &seed)?, + snapshot, + body, + }; + intent.encoded()?; + Ok(Self { intent }) + } + pub fn evidence(&self) -> &PendingMutation { + self.intent.snapshot.evidence() + } + #[cfg(test)] + pub(super) fn command_for_test( + &self, + client: &CellClient, + ) -> cellule_runtime::Result> { + client.restore_command::( + self.intent.snapshot.clone(), + self.intent.body.clone(), + ) + } + #[cfg(test)] + pub(super) fn intent_for_test(&self) -> CustodyIntent { + self.intent.clone() + } + pub async fn register( + &self, + client: &CellClient, + identity: MutationIdentity, + ) -> Result { + let header = self.intent.header()?; + let target = self.evidence().target(); + if let Some(saved) = load(client, target, header.key(), Some(header.step)).await? { + return if saved.intent == self.intent { + Ok(saved) + } else { + Err(CustodyError::Context) + }; + } + // The domain pointer proves registration even after the registration + // identity expires. Absence after an uncertain reply proves nothing. + let result = client + .command::(target, identity, self.intent.clone()) + .await; + let saved = load(client, target, header.key(), Some(header.step)).await?; + if let Some(saved) = saved { + if saved.intent == self.intent { + return Ok(saved); + } + return Err(CustodyError::Context); + } + match result { + Err(error) => Err(CustodyError::Registration(Box::new(error))), + Ok(_) => Err(CustodyError::Context), + } + } +} +impl RegisteredCustody { + /// Trusted private service inventory; loading does not grant current Write. + pub async fn load_latest( + client: &CellClient, + target: &CellTarget, + operation: [u8; 16], + ) -> Result, CustodyError> { + load(client, target, operation, None).await + } + pub async fn load_for( + client: &CellClient, + target: &CellTarget, + purpose: CustodyPurpose, + operation: [u8; 16], + ) -> Result, CustodyError> { + load(client, target, CustodyKey { purpose, operation }, None).await + } + pub fn evidence(&self) -> &PendingMutation { + self.intent.snapshot.evidence() + } + pub fn action(&self) -> Result { + Ok(self.intent.request()?.action) + } + pub fn settled(&self) -> bool { + self.phase.is_some() + } + /// Logical closure is separate from an original execution result. + pub fn closed(&self) -> bool { + self.phase.is_some() || self.stopped.is_some() + } + pub fn stop_fact(&self) -> Option { + self.stopped + .as_ref() + .map(|record| record.fact(self.evidence().target())) + } + pub async fn recover_preparation( + &self, + client: &CellClient, + ) -> Result, InvocationError> { + project(self.recover(client).await, |reply| match reply { + CustodyReply::Preparation(reply) => Some(reply), + _ => None, + }) + } + pub async fn recover_serving( + &self, + client: &CellClient, + ) -> Result, InvocationError> { + project(self.recover(client).await, |reply| match reply { + CustodyReply::Serving(reply) => Some(reply), + _ => None, + }) + } + pub async fn recover_staging( + &self, + client: &CellClient, + ) -> Result, InvocationError> { + project(self.recover(client).await, |reply| match reply { + CustodyReply::Staging(reply) => Some(reply), + _ => None, + }) + } + pub async fn recover( + &self, + client: &CellClient, + ) -> Result, InvocationError> { + self.recover_guarded(client, || Ok(())).await + } + async fn recover_guarded( + &self, + client: &CellClient, + before_execute: impl FnOnce() -> Result<(), Error> + Send, + ) -> Result, InvocationError> { + let evidence = self.evidence(); + let recover = async { + let header = self.intent.header()?; + let current = load(client, evidence.target(), header.key(), Some(header.step)) + .await + .map_err(|_| Error::Command("custody phase query failed"))? + .ok_or(Error::Command("custody intent disappeared"))?; + if current.intent != self.intent { + return Err(Error::Command("custody recovery binding differs")); + } + if current.stopped.is_some() { + return Err(Error::Command("custody original retired without execution")); + } + if let Some(phase) = current.phase { + return Ok(Some(phase.committed(evidence)?)); + } + Ok::<_, Error>(None::>) + } + .await; + // Keep original evidence on storage/codec failures, rather than + // misclassifying them as permission to execute or replace the head. + match recover { + Ok(Some(known)) => return normalize(known), + Ok(None) => {} + Err(_) => return Err(InvocationError::Pending(Box::new(evidence.clone()))), + } + if super::exact::known::(client, evidence, 512) + .await? + .is_some() + { + // Atomic receiver publication requires a domain phase whenever SDK + // reports acceptance. Missing application knowledge is corruption. + return Err(InvocationError::Pending(Box::new(evidence.clone()))); + } + before_execute().map_err(InvocationError::NotStarted)?; + let command = client + .restore_command::( + self.intent.snapshot.clone(), + self.intent.body.clone(), + ) + .map_err(InvocationError::NotStarted)?; + normalize(command.execute().await?) + } +} +fn normalize( + committed: Committed, +) -> Result, InvocationError> { + if committed.output.rejected() { + Err(InvocationError::Rejected(Box::new(committed))) + } else { + Ok(committed) + } +} + +pub(super) fn project( + result: Result, InvocationError>, + output: impl FnOnce(CustodyReply) -> Option, +) -> Result, InvocationError> { + let committed = |value: Committed| { + let receipt = value.receipt; + output(value.output) + .map(|output| Committed { output, receipt }) + .ok_or(InvocationError::InvalidPublishedResult { + receipt, + source: Box::new(Error::Command("custody reply purpose differs")), + }) + }; + match result { + Ok(value) => committed(value), + Err(InvocationError::Rejected(value)) => { + Err(InvocationError::Rejected(Box::new(committed(*value)?))) + } + Err(InvocationError::Pending(evidence)) => Err(InvocationError::Pending(evidence)), + Err(InvocationError::NotStarted(error)) => Err(InvocationError::NotStarted(error)), + Err(InvocationError::InvalidPublishedResult { receipt, source }) => { + Err(InvocationError::InvalidPublishedResult { receipt, source }) + } + } +} + +/// A recorded grant can authorize allocating a *new* attempt after the old row +/// was reaped. It never reinstates the old namespace, generation pin or clock. +pub(super) fn restart_matches( + context: &CommandContext<'_, '_>, + check: &LeaseCheck, + staging: bool, +) -> cellule_runtime::Result { + let sets = context.sql(&SqlBatch { statements: vec![SqlStatement { + sql: "SELECT step,incarnation,request_id,intent,phase,stopped FROM catalog_custody_commands INDEXED BY catalog_custody_grants WHERE purpose=0 AND operation=?1 AND granted_incarnation=?2 AND granted_attempt=?3 ORDER BY step DESC LIMIT 1".into(), + parameters: vec![blob(check.token.operation), blob(check.token.owner.incarnation.as_bytes()), number(check.token.attempt)?], + }, seed_statement()] })?; + let Some(saved) = from_sets(&sets, context.target(), check.token.operation)? else { + return Ok(false); + }; + let header = saved.intent.header()?; + if header.actor != check.actor { + return Ok(false); + } + let Some(phase) = saved.phase else { + return Err(Error::Command("custody restart grant is unsettled")); + }; + Ok(match phase.decode_reply::()? { + CustodyReply::Preparation(PreparationReply::Granted(lease)) if !staging => { + lease.token == check.token + } + CustodyReply::Staging(StagingReply::Granted(lease)) if staging => { + lease.token == check.token + } + _ => false, + }) +} diff --git a/crates/canopy-server/src/packs/publication/custody/scan.rs b/crates/canopy-server/src/packs/publication/custody/scan.rs new file mode 100644 index 00000000..be478e76 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/custody/scan.rs @@ -0,0 +1,228 @@ +//! Bounded keyset discovery; existing maintenance admission owns exact stops. +use super::super::scan::ScanControl; +use super::*; +use std::sync::Arc; +use tokio::{sync::watch, task::JoinHandle}; + +const SEEK: &str = "SELECT purpose,operation FROM catalog_custody_commands INDEXED BY catalog_custody_pending WHERE phase IS NULL AND stopped IS NULL AND (purpose,operation)>(?1,?2) ORDER BY purpose,operation LIMIT ?3"; +#[derive(Clone, Debug, Default)] +pub struct CustodyScanStats { + pub passes: u64, + pub scanned: u64, + pub submitted: u64, + pub recovered: u64, + pub deferred: u64, + pub failures: u64, + pub last_error: Option>, +} +impl CustodyScanStats { + fn failed(&mut self, error: CustodyError) { + self.failures = self.failures.saturating_add(1); + self.last_error = Some(Arc::new(error)); + } +} +#[must_use] +pub struct CustodySupervisor { + control: ScanControl, + stats: watch::Receiver, + task: Option>, +} +impl CustodySupervisor { + pub fn start( + client: CellClient, + target: CellTarget, + coordinator: PublicationCoordinator, + settings: RecoveryScanSettings, + authority: PreparationAuthority, + ) -> Result { + if settings.validate().is_err() { + return Err(CustodyError::InvalidScanLimits); + } + if !authority.matches(&target) || coordinator.target() != &target { + return Err(CustodyError::Context); + } + let sql = SqlCell::::new(client.clone(), target.clone())?; + let control = ScanControl::default(); + let (updates, stats) = watch::channel(CustodyScanStats::default()); + let scan = Scan { + client, + target, + coordinator, + authority, + }; + let task = settings.spawn(run(scan, sql, settings.clone(), control.clone(), updates)); + Ok(Self { + control, + stats, + task: Some(task), + }) + } + pub(crate) async fn pause(&self) { + self.control.pause().await; + } + pub(crate) fn resume(&self) { + self.control.resume(); + } + pub fn stats(&self) -> CustodyScanStats { + self.stats.borrow().clone() + } + pub async fn shutdown(mut self) -> Result { + self.control.stop(); + self.task.take().expect("custody scan owner").await + } +} +impl Drop for CustodySupervisor { + fn drop(&mut self) { + self.control.stop(); + } +} +struct Scan { + client: CellClient, + target: CellTarget, + coordinator: PublicationCoordinator, + authority: PreparationAuthority, +} +impl Scan { + async fn visit( + &self, + key: CustodyKey, + stats: &mut CustodyScanStats, + ) -> Result<(), CustodyError> { + // The coordinator owns accepted uncertainty even if its SQL key vanished. + // Do not create a second retirement identity for an admitted operation. + if self + .coordinator + .pending_custody_stop_for(key.purpose, key.operation) + .await + .is_some() + { + stats.deferred = stats.deferred.saturating_add(1); + return Ok(()); + } + let Some(saved) = load(&self.client, &self.target, key, None).await? else { + return Ok(()); + }; + if saved.closed() || saved.evidence().identity().expires_at_ms > now(0)? { + stats.deferred = stats.deferred.saturating_add(1); + return Ok(()); + } + let identity = crate::server::mutation_identity() + .map_err(|source| CustodyError::Clock(Box::new(source)))?; + let ready = saved + .ready_stop(self.client.clone(), identity, &self.authority) + .await?; + match self.coordinator.submit(ready).await { + Ok(_) => stats.submitted = stats.submitted.saturating_add(1), + Err(failure) => match failure.reason { + PublicationScheduleError::Capacity + | PublicationScheduleError::Duplicate + | PublicationScheduleError::Closed => { + stats.deferred = stats.deferred.saturating_add(1); + } + _ => return Err(CustodyError::Context), + }, + } + Ok(()) + } +} +async fn page( + sql: &SqlCell, + after: CustodyKey, + count: u16, +) -> Result, CustodyError> { + let result = sql + .query( + None, + statement( + SEEK, + vec![ + number(u64::from(after.purpose.number()))?, + blob(after.operation), + number(u64::from(count))?, + ], + ), + ) + .await + .map_err(|error| CustodyError::Query(Box::new(error)))?; + let rows = rows(&result.output)?; + if rows.len() > count as usize { + return Err(CustodyError::Context); + } + let mut keys = Vec::with_capacity(rows.len()); + let mut previous = after; + for row in rows { + let [SqlValue::Integer(purpose), key] = row.as_slice() else { + return Err(CustodyError::Context); + }; + let key = CustodyKey { + purpose: CustodyPurpose::parse( + u8::try_from(*purpose).map_err(|_| CustodyError::Context)?, + )?, + operation: fixed::<16>(key)?, + }; + if key <= previous { + return Err(CustodyError::Context); + } + keys.push(key); + previous = key; + } + Ok(keys) +} +async fn run( + scan: Scan, + sql: SqlCell, + settings: RecoveryScanSettings, + control: ScanControl, + updates: watch::Sender, +) -> CustodyScanStats { + let mut stats = CustodyScanStats::default(); + let mut after = CustodyKey::from([0; 16]); + loop { + let Some(round) = control.enter(&settings).await else { + return stats; + }; + let permit = match settings.acquire().await { + Ok(permit) => permit, + Err(_) => { + stats.deferred = stats.deferred.saturating_add(1); + updates.send_replace(stats.clone()); + drop(round); + settings.delay(&control).await; + continue; + } + }; + if control.interrupted(&settings) { + drop((permit, round)); + continue; + } + match scan.coordinator.recover_custody_stops().await { + Ok(recovered) => stats.recovered = stats.recovered.saturating_add(recovered), + Err(_) => stats.failed(CustodyError::Context), + } + match page(&sql, after, settings.limits.page).await { + Ok(keys) => { + if keys.is_empty() { + after = CustodyKey::from([0; 16]); + stats.passes = stats.passes.saturating_add(1); + } + for key in keys { + if control.interrupted(&settings) { + break; + } + stats.scanned = stats.scanned.saturating_add(1); + if let Err(error) = scan.visit(key, &mut stats).await { + stats.failed(error); + } + // Advance even for corrupt heads; revisit on the next pass. + after = key; + } + } + Err(error) => stats.failed(error), + } + // Publish the completed round before releasing its quiescence guard: + // a successful pause also makes its diagnostics stable. + updates.send_replace(stats.clone()); + drop((permit, round)); + settings.delay(&control).await; + } +} diff --git a/crates/canopy-server/src/packs/publication/custody/stop.rs b/crates/canopy-server/src/packs/publication/custody/stop.rs new file mode 100644 index 00000000..246879e8 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/custody/stop.rs @@ -0,0 +1,491 @@ +//! Retire an expired original without inventing its execution result or receipt. +use super::*; +use cellule_runtime::{PreparedCommand, Receipt}; + +const DOMAIN: &[u8] = b"canopy.custody-retirement.v2\0"; +#[derive(Clone, Debug, PartialEq, Eq)] +struct StopData { + purpose: CustodyPurpose, + tenant: [u8; 16], + application: [u8; 16], + operation: [u8; 16], + step: u32, + intent_digest: [u8; 32], + owner: OwnerFence, +} +impl WireValue for StopData { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.operation == [0; 16] || self.step > MAX_STEPS || self.owner.epoch == 0 { + return Err(CodecError::Invalid("custody retirement context")); + } + e.write_bytes(DOMAIN)?; + e.write_u8(self.purpose.number())?; + e.write_bytes(&self.tenant)?; + e.write_bytes(&self.application)?; + e.write_bytes(&self.operation)?; + e.write_u32(self.step)?; + e.write_bytes(&self.intent_digest)?; + e.write_bytes(self.owner.incarnation.as_bytes())?; + e.write_u64(self.owner.epoch) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("custody retirement purpose")); + } + let value = Self { + purpose: CustodyPurpose::parse(d.read_u8()?)?, + tenant: crate::packs::directory::index::codec::fixed(d)?, + application: crate::packs::directory::index::codec::fixed(d)?, + operation: crate::packs::directory::index::codec::fixed(d)?, + step: d.read_u32()?, + intent_digest: crate::packs::directory::index::codec::fixed(d)?, + owner: OwnerFence { + incarnation: IncarnationId::from_bytes( + crate::packs::directory::index::codec::fixed(d)?, + ), + epoch: d.read_u64()?, + }, + }; + value.encode(&mut BoundedEncoder::new(512)?)?; + Ok(value) + } +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CustodyStopInput { + certificate: CertificateEnvelope, +} +impl WireValue for CustodyStopInput { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.certificate.data::()?; + self.certificate.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + certificate: CertificateEnvelope::decode(d)?, + }; + value.encode(&mut BoundedEncoder::new(1024)?)?; + Ok(value) + } +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CustodyStopReply { + Stopped, + Settled, + Denied(PreparationDenial), +} +impl WireValue for CustodyStopReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Stopped => e.write_u8(0), + Self::Settled => e.write_u8(1), + Self::Denied(reason) => { + e.write_u8(2)?; + PreparationReply::Denied(*reason).encode(e) + } + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 0 => Ok(Self::Stopped), + 1 => Ok(Self::Settled), + 2 => match PreparationReply::decode(d)? { + PreparationReply::Denied(reason) => Ok(Self::Denied(reason)), + _ => Err(CodecError::Invalid("grant in custody retirement denial")), + }, + _ => Err(CodecError::Invalid("custody retirement reply")), + } + } +} +/// Logical closure of an original, never that original's SDK outcome. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CustodyStopFact { + pub receipt: Receipt, + pub owner: OwnerFence, + pub stopped_at_ms: i64, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct StopRecord { + data: StopData, + stamp: Stamp, + stopped_at_ms: i64, + result: Recorded, +} +impl WireValue for StopRecord { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.stopped_at_ms < 0 + || self.result.rejected() + || self.result.decode_reply::()? != CustodyStopReply::Stopped + { + return Err(CodecError::Invalid("custody retirement result")); + } + self.data.encode(e)?; + self.stamp.encode(e)?; + e.write_i64(self.stopped_at_ms)?; + self.result.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + data: StopData::decode(d)?, + stamp: Stamp::decode(d)?, + stopped_at_ms: d.read_i64()?, + result: Recorded::decode(d)?, + }; + value.encode(&mut BoundedEncoder::new(960)?)?; + Ok(value) + } +} +impl StopRecord { + pub(super) fn fact(&self, target: &CellTarget) -> CustodyStopFact { + CustodyStopFact { + receipt: Receipt { + cell: target.cell_id(), + incarnation: self.data.owner.incarnation, + commit_sequence: self.result.sequence(), + }, + owner: self.data.owner, + stopped_at_ms: self.stopped_at_ms, + } + } +} +pub(super) fn record( + value: &SqlValue, + intent: &CustodyIntent, + seed: &[u8; 32], +) -> cellule_runtime::Result> { + let bytes = match value { + SqlValue::Null => return Ok(None), + SqlValue::Blob(bytes) => bytes, + _ => return Err(Error::Command("custody retirement row")), + }; + let certificate: CertificateEnvelope = decode(bytes, 1024)?; + if !certificate.authenticated(seed) { + return Err(Error::Command("custody retirement authentication")); + } + let value: StopRecord = certificate.data()?; + let header = intent.header()?; + if value.data.tenant != header.tenant + || value.data.application != header.application + || value.data.purpose != header.purpose + || value.data.operation != header.operation + || value.data.step != header.step + || value.data.intent_digest != *blake3::hash(&intent.encoded()?).as_bytes() + || value.stopped_at_ms < intent.snapshot.evidence().identity().expires_at_ms + { + return Err(Error::Command("custody retirement binding")); + } + Ok(Some(value)) +} + +pub struct StopCustodyIntent; +impl Command for StopCustodyIntent { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 43; + const CODEC_VERSION: u32 = 2; + type Input = CustodyStopInput; + type Output = CustodyStopReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + let deny = |reason| Ok(CommandResult::Rejected(CustodyStopReply::Denied(reason))); + let data: StopData = input.certificate.data()?; + let sets = context.sql(&SqlBatch { + statements: vec![ + row_statement( + CustodyKey { + purpose: data.purpose, + operation: data.operation, + }, + Some(data.step), + ), + seed_statement(), + ], + })?; + let seed = super::super::attestation::seed(&sets[1..])?; + if !input.certificate.authenticated(&seed) + || data.tenant != *context.target().tenant().as_bytes() + || data.application != *context.target().application().as_bytes() + { + return deny(PreparationDenial::Unauthorized); + } + let Some(saved) = from_sets( + &sets, + context.target(), + CustodyKey { + purpose: data.purpose, + operation: data.operation, + }, + )? + else { + return deny(PreparationDenial::Missing); + }; + if data.intent_digest != *blake3::hash(&saved.intent.encoded()?).as_bytes() { + return deny(PreparationDenial::Conflict); + } + // These are logical observations, not grants, and precede current owner + // checks. Neither an accepted original nor an existing stop is rewritten. + if saved.settled() { + return Ok(CommandResult::Success(CustodyStopReply::Settled)); + } + if saved.stopped.is_some() { + return Ok(CommandResult::Success(CustodyStopReply::Stopped)); + } + if data.owner != context.owner_fence() { + return deny(PreparationDenial::Stale); + } + let stopped_at_ms = now(context.now_ms())?; + if saved.evidence().identity().expires_at_ms > stopped_at_ms { + return deny(PreparationDenial::Conflict); + } + let evidence = context + .mutation_evidence() + .ok_or(Error::Command("custody retirement evidence absent"))?; + let result = Recorded::new( + context.sequence(), + false, + encode(&CustodyStopReply::Stopped, 128)?, + )?; + let record = StopRecord { + data, + stamp: Stamp::of(&evidence), + stopped_at_ms, + result, + }; + let bytes = encode(&CertificateEnvelope::seal(&record, &seed)?, 1024)?; + super::super::publish::changed(context.sql(&statement( + "UPDATE catalog_custody_commands SET stopped=?1 WHERE operation=?2 AND step=?3 AND intent=?4 AND phase IS NULL AND stopped IS NULL AND purpose=?5", + vec![SqlValue::Blob(bytes),blob(record.data.operation), number(u64::from(record.data.step))?,SqlValue::Blob(saved.intent.encoded()?),number(u64::from(record.data.purpose.number()))?], + ))?)?; + Ok(CommandResult::Success(CustodyStopReply::Stopped)) + } +} + +#[derive(Clone, Debug)] +pub struct CustodyStopOutcome { + pub purpose: CustodyPurpose, + pub original: PendingMutation, + pub invocation: PendingMutation, + pub stop: Option, + /// None when a different first-writer stop proves logical closure. This + /// deliberately makes no execution claim about this invocation identity. + pub committed: Option>, +} +#[derive(Clone)] +#[must_use] +pub struct ReadyCustodyStop { + purpose: CustodyPurpose, + client: CellClient, + target: CellTarget, + request: BeginRequest, + step: u32, + intent_digest: [u8; 32], + original: PendingMutation, + command: PreparedCommand, +} +impl RegisteredCustody { + pub async fn ready_stop( + &self, + client: CellClient, + identity: MutationIdentity, + authority: &PreparationAuthority, + ) -> Result { + let target = self.evidence().target().clone(); + let header = self.intent.header()?; + let current = load(&client, &target, header.key(), Some(header.step)) + .await? + .ok_or(CustodyError::Context)?; + if current.intent != self.intent { + return Err(CustodyError::Context); + } + if let Some(fact) = current.stop_fact() { + return Err(CustodyError::Stopped(Box::new(fact))); + } + if current.settled() { + return Err(CustodyError::Context); + } + let owner = authority + .observe(&target) + .await + .map_err(|error| CustodyError::Owner(Box::new(error)))?; + let seed = super::super::attestation::seed( + &SqlCell::::new(client.clone(), target.clone())? + .query( + None, + statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ), + ) + .await + .map_err(|error| CustodyError::Query(Box::new(error)))? + .output, + )?; + let intent_digest = *blake3::hash(&self.intent.encoded()?).as_bytes(); + let data = StopData { + purpose: header.purpose, + tenant: header.tenant, + application: header.application, + operation: header.operation, + step: header.step, + intent_digest, + owner, + }; + let input = CustodyStopInput { + certificate: CertificateEnvelope::seal(&data, &seed)?, + }; + input.encode(&mut BoundedEncoder::new(1024)?)?; + let command = client + .prepare_command::(&target, identity, input) + .await + .map_err(|error| CustodyError::StopPreparation(Box::new(error)))?; + authority + .check(&target, owner) + .await + .map_err(|error| CustodyError::Owner(Box::new(error)))?; + Ok(ReadyCustodyStop { + purpose: header.purpose, + client, + target, + request: BeginRequest { + repository: header.repository, + operation: header.operation, + request_digest: header.request_digest, + actor: header.actor, + lease_ms: DEFAULT_LEASE_MS, + }, + step: header.step, + intent_digest, + original: self.evidence().clone(), + command, + }) + } +} +impl ReadyCustodyStop { + pub(in crate::packs::publication) fn purpose(&self) -> CustodyPurpose { + self.purpose + } + pub fn evidence(&self) -> &PendingMutation { + self.command.evidence() + } + pub fn original(&self) -> &PendingMutation { + &self.original + } + pub(in crate::packs::publication) fn context( + &self, + ) -> (&CellClient, &CellTarget, BeginRequest) { + (&self.client, &self.target, self.request.clone()) + } + pub(in crate::packs::publication) fn pending(&self) -> PublicationError { + PublicationError::CustodyStop(InvocationError::Pending(Box::new(self.evidence().clone()))) + } + async fn recorded(&self) -> Result, CustodyError> { + let saved = load( + &self.client, + &self.target, + CustodyKey { + purpose: self.purpose, + operation: self.request.operation, + }, + Some(self.step), + ) + .await? + .ok_or(CustodyError::Context)?; + if *blake3::hash(&saved.intent.encoded()?).as_bytes() != self.intent_digest { + return Err(CustodyError::Context); + } + Ok(saved.stopped) + } + /// The repository's admitted, cancellation-owned cold transition may retire + /// its one initialization head without starting a separate discovery task. + /// This does not authorize native work or discard the original's evidence. + pub(crate) async fn complete_tracked(self) -> Result { + self.complete_exact(false, 0).await + } + pub(in crate::packs::publication) async fn dispatch( + self, + recover: bool, + fault: u8, + ) -> Result { + self.complete_exact(recover, fault) + .await + .map(|outcome| PublicationOutcome::CustodyStop(Box::new(outcome))) + } + async fn complete_exact( + self, + recover: bool, + fault: u8, + ) -> Result { + let saved = self + .recorded() + .await + .map_err(|source| PublicationError::Custody { + evidence: Box::new(self.evidence().clone()), + source: Box::new(source), + })?; + let invocation = self.evidence().clone(); + let (stop, committed) = if let Some(saved) = saved { + let committed = if saved.stamp == Stamp::of(&invocation) + && saved.data.owner.incarnation == invocation.incarnation() + { + Some(saved.result.committed(&invocation).map_err(|source| { + PublicationError::Custody { + evidence: Box::new(invocation.clone()), + source: Box::new(source.into()), + } + })?) + } else { + None + }; + (Some(saved.fact(&self.target)), committed) + } else { + let committed = super::super::exact::invoke( + &self.client, + self.command.clone(), + recover, + 128, + fault, + ) + .await + .map_err(PublicationError::CustodyStop)?; + let saved = self + .recorded() + .await + .map_err(|source| PublicationError::Custody { + evidence: Box::new(invocation.clone()), + source: Box::new(source), + })?; + if committed.output == CustodyStopReply::Stopped && saved.is_none() { + return Err(self.pending()); + } + (saved.map(|value| value.fact(&self.target)), Some(committed)) + }; + Ok(CustodyStopOutcome { + purpose: self.purpose, + original: self.original, + invocation, + stop, + committed, + }) + } + #[cfg(test)] + pub(in crate::packs::publication) fn input_for_test( + &self, + ) -> Result { + decode(self.command.input_bytes(), 1024) + } + #[cfg(test)] + pub(in crate::packs::publication) fn command_for_test( + &self, + ) -> PreparedCommand { + self.command.clone() + } +} + +#[cfg(test)] +impl CustodyStopInput { + pub(in crate::packs::publication) fn tamper_for_test(mut self) -> Self { + let last = self.certificate.body.len() - 1; + self.certificate.body[last] ^= 2; + self + } +} diff --git a/crates/canopy-server/src/packs/publication/initialization.rs b/crates/canopy-server/src/packs/publication/initialization.rs index 528758f2..97c76e1f 100644 --- a/crates/canopy-server/src/packs/publication/initialization.rs +++ b/crates/canopy-server/src/packs/publication/initialization.rs @@ -1,17 +1,74 @@ //! Fresh empty catalog/ref publication. No SQL-ref conversion or decoded-root //! signing adapter exists: only a privately assembled empty catalog can mint it. use super::*; -use crate::packs::ref_state::{RefSnapshotError, RefStateSnapshot}; +use crate::packs::{ + catalog::CatalogSnapshot, + directory::snapshot::DirectorySnapshot, + ref_state::{RefSnapshotError, RefStateSnapshot}, +}; use cellule_runtime::{InvocationError, primitives::sql::SqlCell}; use tokio::time::timeout_at; -mod publish; +pub(in crate::packs::publication) mod publish; pub use publish::{CheckInitializedCatalog, InitializeCatalogRefs}; pub const INITIALIZATION_BYTES: u32 = 2048; const INITIAL_HEAD: &str = "refs/heads/main"; +#[derive(Debug, thiserror::Error)] +pub enum InitializationVerificationError { + #[error("initialization fact encoding failed")] + Codec(#[from] CodecError), + #[error("initialization catalog metadata failed")] + Catalog(#[from] crate::packs::directory::index::IndexError), + #[error("initialization ref metadata failed")] + Refs(#[from] RefSnapshotError), + #[error("initialization is not the certified empty state")] + Context, +} + +/// Verify the complete constant-size initial graph, including its typed empty +/// leaves. Both route activation and retirement use the same checks. +pub(crate) async fn verify_empty( + fact: GenerationFact, + store: &canopy_object_storage::artifact::ArtifactStore, + format: ObjectFormat, +) -> Result< + ( + StoredCatalog, + crate::packs::directory::snapshot::StoredSnapshot, + RefStateSnapshotRoot, + ), + InitializationVerificationError, +> { + initial_fact(&fact)?; + let catalog = fact + .catalog + .ok_or(InitializationVerificationError::Context)?; + let refs = fact.refs.ok_or(InitializationVerificationError::Context)?; + if catalog.repository != store.repository() || catalog.format != format { + return Err(InitializationVerificationError::Context); + } + let snapshot = CatalogSnapshot::download(store, catalog).await?; + let directory = DirectorySnapshot::download(store, snapshot.directory).await?; + let state = refs.read(store).await?; + if snapshot.sources.is_some() + || !directory.level_zero.is_empty() + || directory.levels.iter().any(Option::is_some) + || state.repository != store.repository() + || state.format != format + || state.generation != 0 + || state.root.is_some() + || state.default_branch != INITIAL_HEAD + { + return Err(InitializationVerificationError::Context); + } + Ok((catalog, snapshot.directory, refs)) +} + #[derive(Debug, thiserror::Error)] pub enum InitializationPreparationError { + #[error("initialization command preparation failed")] + Command(#[source] Box>), #[error("initialization preparation is inactive")] Base(#[from] PreparationBaseError), #[error("initialization snapshot failed")] diff --git a/crates/canopy-server/src/packs/publication/initialization/publish.rs b/crates/canopy-server/src/packs/publication/initialization/publish.rs index 9bd47778..686c49b6 100644 --- a/crates/canopy-server/src/packs/publication/initialization/publish.rs +++ b/crates/canopy-server/src/packs/publication/initialization/publish.rs @@ -20,20 +20,50 @@ fn verification( hash.update(&e.finish()); Ok(*hash.finalize().as_bytes()) } +pub(in crate::packs::publication) const SAVED: &str = "SELECT actor,request_digest,verification_digest,result FROM catalog_initialization WHERE id=?1"; + +pub(in crate::packs::publication) fn selected( + outcome: &[SqlResultSet], + check: &LeaseCheck, +) -> Result, Error> { + let Some( + [ + SqlValue::Text(actor), + request, + digest, + SqlValue::Blob(bytes), + ], + ) = rows(outcome)?.first().map(Vec::as_slice) + else { + if rows(outcome)?.is_empty() { + return Ok(None); + } + return Err(Error::Command("invalid initialization lookup")); + }; + if *actor != check.actor || fixed::<32>(request)? != check.token.request_digest { + return Ok(None); + } + Ok(Some(saved( + bytes, + check.token.repository, + None, + fixed(digest)?, + )?)) +} + fn saved( bytes: &[u8], repository: [u8; 16], - format: ObjectFormat, + format: Option, digest: [u8; 32], ) -> Result { let mut d = BoundedDecoder::new(bytes, 512)?; let fact = GenerationFact::decode(&mut d)?; d.finish()?; initial_fact(&fact)?; - if fact - .catalog - .is_none_or(|root| root.repository != repository || root.format != format) - { + if fact.catalog.is_none_or(|root| { + root.repository != repository || format.is_some_and(|format| root.format != format) + }) { return Err(CodecError::Invalid("invalid initialization result")); } if verification( @@ -53,132 +83,149 @@ pub struct InitializeCatalogRefs; impl Command for InitializeCatalogRefs { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 31; - const CODEC_VERSION: u32 = 1; + const CODEC_VERSION: u32 = 3; type Input = InitialRefProof; type Output = InitializationReply; fn execute( context: &mut CommandContext<'_, '_>, input: Self::Input, ) -> cellule_runtime::Result> { - input.shape()?; - let Some((data, key)) = authenticate( - context, - &input.certificate, - Some(binding(input.refs)?), - None, - )? - else { - return Ok(denied(PreparationDenial::Unauthorized)); + let data = input.certificate.data()?; + let check = LeaseCheck { + token: data.token, + actor: data.actor, }; - let logical = context.sql(&statement("SELECT actor,request_digest,verification_digest,result FROM catalog_initialization WHERE id=?1", vec![blob(data.token.operation)]))?; - if let Some(row) = rows(&logical)?.first() { - let [ - SqlValue::Text(actor), - request, - digest, - SqlValue::Blob(bytes), - ] = row.as_slice() - else { - return Err(Error::Command("invalid initialization outcome")); - }; - if *actor != data.actor - || fixed::<32>(request)? != data.token.request_digest - || fixed::<32>(digest)? != verification(data.catalog, input.refs)? - { - return Ok(denied(PreparationDenial::Conflict)); - } - return Ok(CommandResult::Success(InitializationReply::Initialized( - Box::new(saved( - bytes, - data.token.repository, - data.catalog.format, - fixed(digest)?, - )?), - ))); - } - // Exact recorded replay above grants no write. New initialization must - // satisfy the current actual fence, admin role, pin and pristine state. - if data.token.owner != context.owner_fence() { - return Ok(denied(PreparationDenial::Stale)); - } - let Some(format) = authorized( - context, - data.token.repository, - &data.actor, - TokenScope::Admin, - )? + recovery::execute(context, &check, recovery::Kind::Initialization, |context| { + initialize(context, input) + }) + } +} + +fn initialize( + context: &mut CommandContext<'_, '_>, + input: InitialRefProof, +) -> cellule_runtime::Result> { + input.shape()?; + let Some((data, key)) = authenticate( + context, + &input.certificate, + Some(binding(input.refs)?), + None, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let logical = context.sql(&statement("SELECT actor,request_digest,verification_digest,result FROM catalog_initialization WHERE id=?1", vec![blob(data.token.operation)]))?; + if let Some(row) = rows(&logical)?.first() { + let [ + SqlValue::Text(actor), + request, + digest, + SqlValue::Blob(bytes), + ] = row.as_slice() else { - return Ok(denied(PreparationDenial::Unauthorized)); - }; - let Some(row) = load(context, data.token)? else { - return Ok(denied(PreparationDenial::Missing)); + return Err(Error::Command("invalid initialization outcome")); }; - if !matched( - &row, - &LeaseCheck { - token: data.token, - actor: data.actor.clone(), - }, - ) { - return Ok(denied(PreparationDenial::Stale)); - } - if row.expires <= now(context.now_ms())? { - return Ok(denied(PreparationDenial::Expired)); - } - check_pin(context, &row)?; - if format != data.catalog.format - || !retention_matches(context, &data, row.generation, format)? - || fact(context, data.token.repository, format, None)? != data.base + if *actor != data.actor + || fixed::<32>(request)? != data.token.request_digest + || fixed::<32>(digest)? != verification(data.catalog, input.refs)? { return Ok(denied(PreparationDenial::Conflict)); } - let pristine = context.sql(&statement("SELECT generation=0 AND default_branch=?1 AND NOT EXISTS(SELECT 1 FROM refs) AND NOT EXISTS(SELECT 1 FROM pushes WHERE initial_staging IS NULL OR response_id IS NOT NULL OR publication IS NOT NULL) AND NOT EXISTS(SELECT 1 FROM catalog_compactions) AND NOT EXISTS(SELECT 1 FROM catalog_initialization) AND NOT EXISTS(SELECT 1 FROM catalog_generations WHERE generation>0) FROM ref_generation WHERE singleton=1", vec![SqlValue::Text(INITIAL_HEAD.into())]))?; - match rows(&pristine)?.first().map(Vec::as_slice) { - Some([SqlValue::Integer(1)]) => {} - Some([SqlValue::Integer(0)]) => return Ok(denied(PreparationDenial::Conflict)), - _ => return Err(Error::Command("missing initialization state")), - } - let Some(missing) = checkpoint(context, &data, &key)? else { - return Ok(denied(PreparationDenial::Conflict)); - }; - let verified_roots = verification(data.catalog, input.refs)?; - let bytes = input.certificate.bytes()?; - let digest = *blake3::hash(&bytes).as_bytes(); - let result = GenerationFact { - generation: 1, - catalog: Some(data.catalog), - refs: Some(input.refs), - certificate: Some(digest), - }; - let mut encoded = BoundedEncoder::new(512)?; - result.encode(&mut encoded)?; - let mut catalog = BoundedEncoder::new(256)?; - data.catalog.encode(&mut catalog)?; - let mut refs = BoundedEncoder::new(128)?; - input.refs.encode(&mut refs)?; - if row.expires <= now(context.now_ms())? { - return Ok(denied(PreparationDenial::Expired)); - } - // No rejection after the first write: later failures abort all roots, - // the durable outcome and checkpoint together at the Cell ACK gate. - if missing { - changed(context.sql(&statement("UPDATE catalog_operations SET attestation=?1,attestation_digest=?2 WHERE id=?3 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.operation)]))?)?; - changed(context.sql(&statement("UPDATE catalog_leases SET attestation=?1,attestation_digest=?2 WHERE incarnation=?3 AND admission_sequence=?4 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?]))?)?; - } - changed(context.sql(&statement("INSERT INTO catalog_generations(generation,catalog,certificate,refs) VALUES(1,?1,?2,?3)", vec![blob(catalog.finish()),blob(digest),blob(refs.finish())]))?)?; - changed(context.sql(&statement( - "UPDATE catalog_state SET generation=1 WHERE singleton=1 AND generation=0", - vec![], - ))?)?; - changed(context.sql(&statement("INSERT INTO catalog_initialization(singleton,id,actor,request_digest,verification_digest,result) VALUES(1,?1,?2,?3,?4,?5)", vec![blob(data.token.operation),SqlValue::Text(data.actor),blob(data.token.request_digest),blob(verified_roots),blob(encoded.finish())]))?)?; - changed(context.sql(&statement( - "DELETE FROM catalog_operations WHERE id=?1", - vec![blob(data.token.operation)], - ))?)?; - Ok(CommandResult::Success(InitializationReply::Initialized( - Box::new(result), - ))) + return Ok(CommandResult::Success(InitializationReply::Initialized( + Box::new(saved( + bytes, + data.token.repository, + Some(data.catalog.format), + fixed(digest)?, + )?), + ))); + } + // Exact recorded replay above grants no write. New initialization must + // satisfy the current actual fence, admin role, pin and pristine state. + if data.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(format) = authorized( + context, + data.token.repository, + &data.actor, + TokenScope::Admin, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let Some(row) = load(context, data.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: data.token, + actor: data.actor.clone(), + }, + ) { + return Ok(denied(PreparationDenial::Stale)); + } + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + check_pin(context, &row)?; + if format != data.catalog.format + || !retention_matches(context, &data, row.generation, format)? + || fact(context, data.token.repository, format, None)? != data.base + { + return Ok(denied(PreparationDenial::Conflict)); + } + let pristine = context.sql(&statement("SELECT generation=0 AND default_branch=?1 AND NOT EXISTS(SELECT 1 FROM refs) AND NOT EXISTS(SELECT 1 FROM pushes WHERE (initial_staging IS NULL AND initial_preparation IS NULL) OR response_id IS NOT NULL OR publication IS NOT NULL) AND NOT EXISTS(SELECT 1 FROM catalog_compactions) AND NOT EXISTS(SELECT 1 FROM catalog_initialization) AND NOT EXISTS(SELECT 1 FROM catalog_generations WHERE generation>0) FROM ref_generation WHERE singleton=1", vec![SqlValue::Text(INITIAL_HEAD.into())]))?; + match rows(&pristine)?.first().map(Vec::as_slice) { + Some([SqlValue::Integer(1)]) => {} + Some([SqlValue::Integer(0)]) => return Ok(denied(PreparationDenial::Conflict)), + _ => return Err(Error::Command("missing initialization state")), + } + let Some(missing) = checkpoint(context, &data, &key)? else { + return Ok(denied(PreparationDenial::Conflict)); + }; + let verified_roots = verification(data.catalog, input.refs)?; + let bytes = input.certificate.bytes()?; + let digest = *blake3::hash(&bytes).as_bytes(); + let result = GenerationFact { + generation: 1, + catalog: Some(data.catalog), + refs: Some(input.refs), + certificate: Some(digest), + }; + let mut encoded = BoundedEncoder::new(512)?; + result.encode(&mut encoded)?; + let mut catalog = BoundedEncoder::new(256)?; + data.catalog.encode(&mut catalog)?; + let mut refs = BoundedEncoder::new(128)?; + input.refs.encode(&mut refs)?; + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + // No rejection after the first write: later failures abort all roots, + // the durable outcome and checkpoint together at the Cell ACK gate. + if missing { + changed(context.sql(&statement("UPDATE catalog_operations SET attestation=?1,attestation_digest=?2 WHERE id=?3 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.operation)]))?)?; + changed(context.sql(&statement("UPDATE catalog_leases SET attestation=?1,attestation_digest=?2 WHERE incarnation=?3 AND admission_sequence=?4 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?]))?)?; } + changed(context.sql(&statement( + "INSERT INTO catalog_generations(generation,catalog,certificate,refs) VALUES(1,?1,?2,?3)", + vec![blob(catalog.finish()), blob(digest), blob(refs.finish())], + ))?)?; + changed(context.sql(&statement( + "UPDATE catalog_state SET generation=1 WHERE singleton=1 AND generation=0", + vec![], + ))?)?; + changed(context.sql(&statement("INSERT INTO catalog_initialization(singleton,id,actor,request_digest,verification_digest,result,incarnation,admission_sequence) VALUES(1,?1,?2,?3,?4,?5,?6,?7)", vec![blob(data.token.operation),SqlValue::Text(data.actor),blob(data.token.request_digest),blob(verified_roots),blob(encoded.finish()),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?]))?)?; + changed(context.sql(&statement( + "DELETE FROM catalog_operations WHERE id=?1", + vec![blob(data.token.operation)], + ))?)?; + Ok(CommandResult::Success(InitializationReply::Initialized( + Box::new(result), + ))) } /// Fresh read authorization and exact logical identity; no namespace allocation @@ -232,7 +279,7 @@ impl Query for CheckInitializedCatalog { Ok(Some(saved( bytes, input.repository, - format, + Some(format), fixed(digest)?, )?)) } diff --git a/crates/canopy-server/src/packs/publication/mod.rs b/crates/canopy-server/src/packs/publication/mod.rs index f85a9a3e..ff422d05 100644 --- a/crates/canopy-server/src/packs/publication/mod.rs +++ b/crates/canopy-server/src/packs/publication/mod.rs @@ -1,6 +1,6 @@ //! Fenced preparation and retained generation facts in the Repository Cell. -//! The fresh schema is selected with the final producer/reader hard cutover; -//! these commands are not registered on the legacy repository serving path. +//! Production and qualification share the same bounded packed operation registry. +//! The remaining producer/reader conversion is an unreleasable local cutover. use super::{catalog::StoredCatalog, ref_state::RefStateSnapshotRoot}; use crate::{ ObjectFormat, RepositoryModule, @@ -14,6 +14,20 @@ use cellule_runtime::{ primitives::sql::{SqlBatch, SqlResultSet, SqlStatement, SqlValue}, registry::{CommandContext, CommandResult, OwnerFence, QueryContext}, }; +mod serving; +pub use serving::{ + AcquireServingPin, AcquireServingRequest, CheckServingPin, MAX_EDGE_PARENTS, + MAX_SERVING_GENERATIONS, MAX_SERVING_OWNERS, MAX_SERVING_PINS, NativeWorkspace, + ReadyServingCommand, ReadyServingRelease, ReleaseServingPin, RenewServingPin, + RenewServingRequest, ResolvedServingRef, SelectServingGeneration, ServingCheck, ServingContext, + ServingDenial, ServingDrainObserver, ServingDrainProof, ServingEdgePage, ServingLease, + ServingOwner, ServingOwnerError, ServingOwnerPhase, ServingOwnerStats, ServingPin, ServingPool, + ServingPoolLimits, ServingReadBudget, ServingReadError, ServingReleaseReply, ServingReply, + ServingSelection, ServingSnapshot, ServingToken, WorkspaceLimits, WorkspaceStats, +}; +mod owner; +pub(crate) mod registry; +pub use owner::PreparationAuthority; mod session; pub use session::PreparationSession; mod base; @@ -30,9 +44,10 @@ pub use prepare::{CatalogPreparation, CatalogPreparationError, PreparedCatalog}; pub(in crate::packs) mod ref_proof; pub use ref_proof::{RefProofError, RefPublicationProof}; mod initialization; +pub(crate) use initialization::verify_empty as verify_initial_catalog; pub use initialization::{ CheckInitializedCatalog, INITIALIZATION_BYTES, InitialRefProof, InitializationPreparationError, - InitializationReply, InitializeCatalogRefs, + InitializationReply, InitializationVerificationError, InitializeCatalogRefs, }; mod ref_snapshot; pub use ref_snapshot::{PreparedRefSnapshot, RefSnapshotPreparationError}; @@ -50,6 +65,7 @@ mod outcome; pub use outcome::OutcomeCertificate; mod completion; mod coordinator; +mod scan; pub use completion::{ CatalogCompletionReply, CatalogPushCompletion, CatalogPushResponseError, CheckCompletedPush, CompleteCatalogPush, CompletedCatalogPush, CompletionCatalogProof, PushCompletionProofError, @@ -57,13 +73,15 @@ pub use completion::{ }; pub use coordinator::{ CompactionReadyError, NativeInputReadyError, PreparationCommandKind, PreparationCommandOutcome, - PreparationReadyError, PublicationAdmissionFailure, PublicationClass, PublicationCoordinator, - PublicationError, PublicationLimits, PublicationOutcome, PublicationScheduleError, - PublicationState, PublicationStats, PublicationTicket, ReadyBoundRecovery, - ReadyCatalogCompaction, ReadyCatalogPush, ReadyNativeInputs, ReadyPreparation, - ReadyPublication, ReadyRefPolicyPage, ReadyRootPush, RecoveryBindingFailure, - RefPolicyReadyError, RefPolicyRefusalFailure, RegisteredNativeInputs, RootPushReadyError, + PreparationReadyError, PublicationAdmissionFailure, PublicationBudget, PublicationBudgetStats, + PublicationClass, PublicationCoordinator, PublicationError, PublicationLimits, + PublicationOutcome, PublicationScheduleError, PublicationState, PublicationStats, + PublicationTicket, ReadyBoundRecovery, ReadyCatalogCompaction, ReadyCatalogPush, + ReadyInitialization, ReadyNativeInputs, ReadyPreparation, ReadyPublication, ReadyRefPolicyPage, + ReadyRootPush, RecoveryBindingFailure, RefPolicyReadyError, RefPolicyRefusalFailure, + RegisteredNativeInputs, RootPushReadyError, ServingDrainAdmission, }; +pub use scan::{RecoveryScanBudget, RecoveryScanSettings}; mod commands; mod compaction; pub use compaction::{ @@ -96,6 +114,16 @@ pub use root_completion::{ RootOutcomeCompletion, RootPushCompletion, RootPushOutcomes, RootPushReplayError, RootSignedPushFact, replay_root_push_response, }; +mod admission_receipt; +mod custody; +pub use custody::{ + CustodyAction, CustodyError, CustodyIntent, CustodyPurpose, CustodyReply, CustodyRequest, + CustodyScanStats, CustodyStopFact, CustodyStopInput, CustodyStopOutcome, CustodyStopReply, + CustodySupervisor, ExecuteCustody, PreparedCustody, ReadyCustodyStop, RegisterCustodyIntent, + RegisteredCustody, StopCustodyIntent, +}; +mod preparation_receipt; +pub use preparation_receipt::{PreparationAdmission, PreparationReceiptError}; mod staging_receipt; pub use staging_receipt::{StagingAdmission, StagingReceiptError}; mod staging; @@ -113,7 +141,8 @@ pub use commands::{ pub const SCHEMA: &str = concat!( include_str!("schema.sql"), - include_str!("ref_policy/schema.sql") + include_str!("ref_policy/schema.sql"), + include_str!("serving/schema.sql") ); pub const MAX_OPERATIONS: u64 = 1024; pub const MAX_GENERATION_LEASES: u64 = 4096; @@ -214,21 +243,18 @@ pub struct MaintenanceRequest { pub owner: OwnerFence, } -/// Register on the fresh RepositoryModule only, with bounded descriptors for -/// command IDs 11..14/16..19/22/24..26/28..29/31/33/35 and query IDs -/// 15/20..21/23/27/30/32/34, plus the existing trusted SQL query. No separate -/// Cell or compatibility API. +/// Bind the packed production contract. Inline publication/completion adapters +/// are deliberately excluded; qualification binds its historical fixtures itself. pub fn register(registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_query::()?; + registry.bind_query::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_query::()?; registry.bind_query::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; @@ -237,8 +263,6 @@ pub fn register(registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { registry.bind_command::()?; registry.bind_query::()?; registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; @@ -246,7 +270,6 @@ pub fn register(registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { registry.bind_query::()?; registry.bind_command::()?; registry.bind_query::()?; - registry.bind_query::()?; registry.bind_query::()?; registry.bind_query::() } diff --git a/crates/canopy-server/src/packs/publication/owner.rs b/crates/canopy-server/src/packs/publication/owner.rs new file mode 100644 index 00000000..2270aeee --- /dev/null +++ b/crates/canopy-server/src/packs/publication/owner.rs @@ -0,0 +1,76 @@ +//! Fresh durable owner observations, separate from historical lease replies. +use super::*; +use cellule_runtime::CellTarget; + +/// Server-owned authority source for one repository Cell. A decoded lease or +/// caller-supplied fence cannot construct this capability. +#[derive(Clone)] +pub struct PreparationAuthority { + target: CellTarget, + source: Source, +} +#[derive(Clone)] +enum Source { + Node(crate::server::peer::NodePeer), + // The local runtime fixture has real durable Control ownership but no + // network node advertisement. This path is absent in production builds. + #[cfg(test)] + Local(std::sync::Arc), +} +impl PreparationAuthority { + pub(crate) fn node(peer: crate::server::peer::NodePeer, target: CellTarget) -> Self { + Self { + target, + source: Source::Node(peer), + } + } + #[cfg(test)] + pub(crate) fn local(layout: cellule_ltx::CellStorageLayout, target: CellTarget) -> Self { + Self { + target, + source: Source::Local(std::sync::Arc::new( + cellule_runtime::control::authority::CellAuthority::new(layout), + )), + } + } + pub(super) fn matches(&self, target: &CellTarget) -> bool { + self.target == *target + } + pub(super) async fn check( + &self, + target: &CellTarget, + expected: OwnerFence, + ) -> Result<(), PreparationBaseError> { + if self.observe(target).await? != expected { + return Err(PreparationBaseError::Inactive); + } + Ok(()) + } + pub(super) async fn observe( + &self, + target: &CellTarget, + ) -> Result { + if !self.matches(target) { + return Err(PreparationBaseError::Context); + } + let actual = match &self.source { + Source::Node(peer) => peer + .current_owner_fence(target) + .await + .map_err(|error| PreparationBaseError::Owner(Box::new(error)))?, + #[cfg(test)] + Source::Local(authority) => { + let control = authority + .load(target.cell_id()) + .await + .map_err(|error| PreparationBaseError::Owner(Box::new(error)))? + .ok_or(PreparationBaseError::Inactive)?; + if control.value().owner.is_none() { + return Err(PreparationBaseError::Inactive); + } + control.value().owner_fence() + } + }; + Ok(actual) + } +} diff --git a/crates/canopy-server/src/packs/publication/preparation_receipt.rs b/crates/canopy-server/src/packs/publication/preparation_receipt.rs new file mode 100644 index 00000000..b9c7be33 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/preparation_receipt.rs @@ -0,0 +1,110 @@ +//! First accepted catalog preparation, retained separately from current custody. +pub use super::admission_receipt::AdmissionReceiptError as PreparationReceiptError; +use super::{ + admission_receipt::{Admission, InitialAdmission}, + recovery::phase::Recorded, + *, +}; +use cellule_runtime::{CellClient, CellTarget, Committed, MutationIdentity, PendingMutation}; + +#[derive(Clone)] +pub(super) struct Preparation; +impl Admission for Preparation { + const DOMAIN: &'static [u8] = b"canopy.initial-preparation-receipt.v1\0"; + const COLUMN: &'static str = "initial_preparation"; + type Lease = PreparationLease; + type Reply = PreparationReply; + fn grant(lease: PreparationLease) -> PreparationReply { + PreparationReply::Granted(Box::new(lease)) + } + fn token(lease: &PreparationLease) -> PreparationToken { + lease.token + } + fn lease(result: &Recorded, request: &BeginRequest) -> Result { + let PreparationReply::Granted(lease) = result.decode_reply()? else { + return Err(CodecError::Invalid( + "initial preparation receipt is not a grant", + )); + }; + // Begin can observe a staging attempt that was already bound, or an + // existing preparation. Its command sequence then follows admission. + if result.rejected() + || lease.token.repository != request.repository + || lease.token.operation != request.operation + || lease.token.request_digest != request.request_digest + || lease.token.attempt > result.sequence() + { + return Err(CodecError::Invalid( + "initial preparation receipt result differs", + )); + } + Ok(*lease) + } +} + +/// Authenticated original Begin knowledge, including its actual SDK sequence. +/// The saved clock never extends a live lease. Claim or a fresh Check is needed +/// before any preparation factory can use this attempt. +#[derive(Clone)] +pub struct PreparationAdmission(InitialAdmission); +impl PreparationAdmission { + pub async fn load( + client: &CellClient, + target: &CellTarget, + operation: [u8; 16], + ) -> Result, PreparationReceiptError> { + Ok(InitialAdmission::load(client, target, operation) + .await? + .map(Self)) + } + pub fn request(&self) -> &BeginRequest { + self.0.request() + } + pub fn lease(&self) -> PreparationLease { + self.0.lease() + } + pub fn receipt(&self) -> cellule_runtime::Receipt { + self.0.receipt() + } + pub async fn ready_claim( + &self, + client: CellClient, + lease_ms: u64, + identity: MutationIdentity, + authority: PreparationAuthority, + ) -> Result { + ReadyPreparation::claim( + client, + self.0.target().clone(), + LeaseRequest { + check: LeaseCheck { + token: self.lease().token, + actor: self.request().actor.clone(), + }, + lease_ms, + }, + identity, + authority, + ) + .await + } + pub fn original( + &self, + evidence: &PendingMutation, + ) -> Result>, PreparationReceiptError> { + self.0.original(evidence) + } +} +pub(super) fn save( + context: &CommandContext<'_, '_>, + request: &BeginRequest, + lease: PreparationLease, +) -> cellule_runtime::Result<()> { + super::admission_receipt::save::(context, request, lease) +} +pub(super) fn restart_matches( + context: &CommandContext<'_, '_>, + check: &LeaseCheck, +) -> cellule_runtime::Result { + super::admission_receipt::restart_matches::(context, check) +} diff --git a/crates/canopy-server/src/packs/publication/recovery/archive.rs b/crates/canopy-server/src/packs/publication/recovery/archive.rs index d28abb61..6178eb3d 100644 --- a/crates/canopy-server/src/packs/publication/recovery/archive.rs +++ b/crates/canopy-server/src/packs/publication/recovery/archive.rs @@ -1,8 +1,8 @@ //! A closed attempt transfers the same recovery certificate/journal into its -//! immutable selected-push row before releasing its independent preparation pin. +//! immutable shared receipt row before releasing its independent preparation pin. use super::*; -const DOMAIN: &[u8] = b"canopy.terminal-recovery-release.v1\0"; +const DOMAIN: &[u8] = b"canopy.terminal-recovery-release.v2\0"; const RELEASE_BYTES: u32 = 4096; #[derive(Clone, Debug, PartialEq, Eq)] pub struct TerminalReleaseCertificate(CertificateEnvelope); @@ -40,6 +40,7 @@ pub(in crate::packs::publication) fn descriptor( e.write_text(match kind { ArtifactKind::InputRoot => "input-root", ArtifactKind::InputBody => "input-body", + ArtifactKind::CatalogNode => "catalog-node", _ => return Err(RootRecoveryError::Context), })?; artifact(&mut e, value)?; @@ -119,13 +120,92 @@ impl WireValue for TerminalReleaseReply { } } -impl phase::Journal { - pub(super) fn terminal( +pub(super) enum Terminal { + Push(Box), + Initialization(InitializationReply), +} +impl Terminal { + fn selected_statement(&self, operation: [u8; 16]) -> SqlStatement { + SqlStatement { + sql: match self { + Self::Push(_) => super::super::root_completion::read::SAVED, + Self::Initialization(_) => super::super::initialization::publish::SAVED, + } + .into(), + parameters: vec![blob(operation)], + } + } + fn matches_selected(&self, result: &[SqlResultSet], check: &LeaseCheck) -> Result { + Ok(match self { + Self::Push(terminal) => { + super::super::root_completion::read::saved( + result, + &check.actor, + check.token.request_digest, + )? == Some(**terminal) + } + Self::Initialization(InitializationReply::Initialized(fact)) => { + super::super::initialization::publish::selected(result, check)? == Some(**fact) + } + // A known negative is the original phase knowledge. A later attempt + // may initialize this logical operation, without rewriting that denial. + Self::Initialization(InitializationReply::Denied(_)) => true, + }) + } + async fn closed_graph( &self, - record: &Record, - ) -> Result, CodecError> { + store: &ArtifactStore, + hash: &mut blake3::Hasher, + ) -> Result<(), RootRecoveryError> { + match self { + Self::Push(terminal) => { + super::super::root_completion::closed_graph(store, terminal.root, hash).await? + } + Self::Initialization(reply) => { + hash.update(&encoded(reply, 512)?); + if let InitializationReply::Initialized(fact) = reply { + let format = fact.catalog.ok_or(RootRecoveryError::Context)?.format; + let (catalog, directory, refs) = + super::super::initialization::verify_empty(**fact, store, format).await?; + descriptor( + hash, + catalog.operation, + ArtifactKind::CatalogNode, + catalog.artifact, + )?; + descriptor( + hash, + directory.operation, + ArtifactKind::CatalogNode, + directory.artifact, + )?; + descriptor( + hash, + refs.operation(), + ArtifactKind::InputRoot, + refs.artifact(), + )?; + } + } + } + Ok(()) + } +} +impl phase::Journal { + pub(super) fn terminal(&self, record: &Record) -> Result, CodecError> { // Validation is required even when only a primary result is selected. self.may_advance(record)?; + if record.kind == Kind::Initialization { + return self + .primary + .as_ref() + .map(|value| { + value + .decode_reply::() + .map(Terminal::Initialization) + }) + .transpose(); + } let result = if record.kind == Kind::Policy { if !self.refused(record)? { return Ok(None); @@ -138,12 +218,24 @@ impl phase::Journal { .map(|value| value.decode_reply::()) .transpose()? { - Some(RootCompletionReply::Completed(value)) => Ok(Some(*value)), + Some(RootCompletionReply::Completed(value)) => Ok(Some(Terminal::Push(value))), _ => Ok(None), } } } impl RegisteredRootRecovery { + pub(super) async fn attempt_closed( + &self, + client: &CellClient, + ) -> Result { + let sql = + SqlCell::::new(client.clone(), self.evidence().target().clone())?; + let result = sql.query(None, statement( + "SELECT 1 FROM catalog_operations WHERE incarnation=?1 AND admission_sequence=?2 LIMIT 1", + vec![blob(self.token().owner.incarnation.as_bytes()), number(self.token().attempt)?], + )).await.map_err(|error| RootRecoveryError::Query(Box::new(error)))?; + Ok(rows(&result.output)?.is_empty()) + } /// Complete typed header/history and selected audit verification precedes /// minting the release proof. A historical/intermediate/unknown capability /// cannot release a pin. This proof authorizes no provider deletion. @@ -162,6 +254,9 @@ impl RegisteredRootRecovery { let terminal = journal .terminal(&self.record)? .ok_or(RootRecoveryError::Context)?; + if !self.attempt_closed(&client).await? { + return Err(RootRecoveryError::Context); + } let sql = SqlCell::::new(client.clone(), self.evidence().target().clone())?; let row = sql @@ -169,10 +264,7 @@ impl RegisteredRootRecovery { None, SqlBatch { statements: vec![ - SqlStatement { - sql: super::super::root_completion::read::SAVED.into(), - parameters: vec![blob(self.token().operation)], - }, + terminal.selected_statement(self.token().operation), SqlStatement { sql: "SELECT push_cert_seed FROM repository_identity WHERE singleton=1" .into(), @@ -187,12 +279,7 @@ impl RegisteredRootRecovery { ) .await .map_err(|error| RootRecoveryError::Query(Box::new(error)))?; - if super::super::root_completion::read::saved( - &row.output, - &self.record.check.actor, - self.token().request_digest, - )? != Some(terminal) - { + if !terminal.matches_selected(&row.output, &self.record.check)? { return Err(RootRecoveryError::Context); } let seed = super::super::attestation::seed( @@ -245,7 +332,7 @@ impl RegisteredRootRecovery { )?; record = next; } - super::super::root_completion::closed_graph(store, terminal.root, &mut hash).await?; + terminal.closed_graph(store, &mut hash).await?; let proof = Proof { recovery: self.certificate.clone(), phase: *blake3::hash(&encoded(&journal, 2048)?).as_bytes(), @@ -344,7 +431,7 @@ pub struct ReleaseTerminalRecovery; impl Command for ReleaseTerminalRecovery { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 40; - const CODEC_VERSION: u32 = 1; + const CODEC_VERSION: u32 = 2; type Input = TerminalReleaseInput; type Output = TerminalReleaseReply; fn execute( @@ -397,21 +484,18 @@ impl Command for ReleaseTerminalRecovery { let Some(terminal) = terminal else { return deny(PreparationDenial::Conflict); }; - let selected = context.sql(&statement( - super::super::root_completion::read::SAVED, - vec![blob(record.check.token.operation)], - ))?; - if super::super::root_completion::read::saved( - &selected, - &record.check.actor, - record.check.token.request_digest, - )? != Some(terminal) - { + let selected = context.sql(&SqlBatch { + statements: vec![terminal.selected_statement(record.check.token.operation)], + })?; + if !terminal.matches_selected(&selected, &record.check)? { return deny(PreparationDenial::Conflict); } + // Closing an older denied initialization must not delete the currently + // claimed attempt of the same logical operation. The exact pin is the + // authority boundary; a successor's independent namespace remains live. let active = context.sql(&statement( - "SELECT 1 FROM catalog_operations WHERE id=?1 LIMIT 1", - vec![blob(record.check.token.operation)], + "SELECT 1 FROM catalog_operations WHERE incarnation=?1 AND admission_sequence=?2 LIMIT 1", + vec![blob(record.check.token.owner.incarnation.as_bytes()), number(record.check.token.attempt)?], ))?; if !rows(&active)?.is_empty() { return deny(PreparationDenial::Conflict); @@ -425,7 +509,7 @@ impl Command for ReleaseTerminalRecovery { stamp: Stamp::of(&evidence), result: phase::Recorded::new(context.sequence(), false, encoded(&reply, 128)?)?, }; - let changed=context.sql(&statement("UPDATE pushes SET recovery=?1,recovery_phase=?2,recovery_release=?3 WHERE id=?4 AND recovery IS NULL AND response_root IS NOT NULL",vec![blob(certificate),blob(saved),blob(encoded(&released,1024)?),blob(record.check.token.operation)]))?; + let changed=context.sql(&statement("INSERT INTO catalog_recovery_receipts(incarnation,admission_sequence,operation,recovery,recovery_phase,recovery_release) VALUES(?1,?2,?3,?4,?5,?6)",vec![blob(record.check.token.owner.incarnation.as_bytes()),number(record.check.token.attempt)?,blob(record.check.token.operation),blob(certificate),blob(saved),blob(encoded(&released,1024)?)]))?; if changed.first().is_none_or(|set| set.rows_affected != 1) { return Err(Error::Command("terminal archive CAS failed")); } @@ -446,9 +530,8 @@ impl RegisteredRootRecovery { ) -> Result, RootRecoveryError> { let sql = SqlCell::::new(client.clone(), target.clone())?; let result=sql.query(None,SqlBatch{statements:vec![ - SqlStatement{sql:"SELECT recovery,recovery_phase FROM pushes WHERE id=?1 AND recovery IS NOT NULL".into(),parameters:vec![blob(check.token.operation)]}, + SqlStatement{sql:"SELECT recovery,recovery_phase FROM catalog_recovery_receipts WHERE incarnation=?1 AND admission_sequence=?2 AND operation=?3".into(),parameters:vec![blob(check.token.owner.incarnation.as_bytes()),number(check.token.attempt)?,blob(check.token.operation)]}, SqlStatement{sql:"SELECT push_cert_seed FROM repository_identity WHERE singleton=1 AND repository_id=?1".into(),parameters:vec![blob(check.token.repository)]}, - SqlStatement{sql:super::super::root_completion::read::SAVED.into(),parameters:vec![blob(check.token.operation)]}, ]}).await.map_err(|error|RootRecoveryError::Query(Box::new(error)))?; let Some([SqlValue::Blob(bytes), saved]) = rows(&result.output)?.first().map(Vec::as_slice) else { @@ -471,12 +554,16 @@ impl RegisteredRootRecovery { let terminal = phase::journal(saved, &record)? .terminal(&record)? .ok_or(RootRecoveryError::Context)?; - if super::super::root_completion::read::saved( - result.output.get(2..).ok_or(RootRecoveryError::Context)?, - &check.actor, - check.token.request_digest, - )? != Some(terminal) - { + let selected = sql + .query( + None, + SqlBatch { + statements: vec![terminal.selected_statement(check.token.operation)], + }, + ) + .await + .map_err(|error| RootRecoveryError::Query(Box::new(error)))?; + if !terminal.matches_selected(&selected.output, check)? { return Err(RootRecoveryError::Context); } let bundle = record.root.read::(store, ROOT_BYTES).await?; @@ -489,6 +576,20 @@ impl RegisteredRootRecovery { } } impl ReadyTerminalRelease { + /// The repository's tracked cold-transition task retains this command + /// through cancellation, just as it retains the initial publication. + pub(crate) async fn complete( + self, + ) -> Result, PublicationError> { + let evidence = Box::new(self.command.evidence().clone()); + let PublicationOutcome::TerminalRelease(result) = self.dispatch(false, 0).await? else { + return Err(PublicationError::Recovery { + evidence, + source: Box::new(RootRecoveryError::Context), + }); + }; + Ok(result) + } #[cfg(test)] pub(in crate::packs::publication) fn with_client_for_test( mut self, @@ -520,8 +621,8 @@ impl ReadyTerminalRelease { .query( None, statement( - "SELECT recovery_release FROM pushes WHERE id=?1", - vec![blob(self.check.token.operation)], + "SELECT recovery_release FROM catalog_recovery_receipts WHERE incarnation=?1 AND admission_sequence=?2 AND operation=?3", + vec![blob(self.check.token.owner.incarnation.as_bytes()), number(self.check.token.attempt)?, blob(self.check.token.operation)], ), ) .await diff --git a/crates/canopy-server/src/packs/publication/recovery/codec.rs b/crates/canopy-server/src/packs/publication/recovery/codec.rs index 4f0afd00..e0e15658 100644 --- a/crates/canopy-server/src/packs/publication/recovery/codec.rs +++ b/crates/canopy-server/src/packs/publication/recovery/codec.rs @@ -27,6 +27,7 @@ impl WireValue for Record { Kind::Publish => 0, Kind::Outcome => 1, Kind::Policy => 2, + Kind::Initialization => 3, })?; self.primary.encode(e)?; e.write_bool(self.refusal.is_some())?; @@ -52,6 +53,7 @@ impl WireValue for Record { 0 => Kind::Publish, 1 => Kind::Outcome, 2 => Kind::Policy, + 3 => Kind::Initialization, _ => return Err(CodecError::Invalid("root recovery command")), }, primary: Stamp::decode(d)?, @@ -86,6 +88,7 @@ impl WireValue for Bundle { Kind::Publish => 0, Kind::Outcome => 1, Kind::Policy => 2, + Kind::Initialization => 3, })?; self.primary.encode(e)?; e.write_bool(self.refusal.is_some())?; @@ -103,6 +106,7 @@ impl WireValue for Bundle { 0 => Kind::Publish, 1 => Kind::Outcome, 2 => Kind::Policy, + 3 => Kind::Initialization, _ => return Err(CodecError::Invalid("unknown recovery kind")), }, primary: SavedCommand::decode(d)?, diff --git a/crates/canopy-server/src/packs/publication/recovery/initialization.rs b/crates/canopy-server/src/packs/publication/recovery/initialization.rs new file mode 100644 index 00000000..ddef2b4b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/recovery/initialization.rs @@ -0,0 +1,90 @@ +//! Exact initialization discovery, receipts and command restoration. +use super::*; + +impl RegisteredRootRecovery { + /// Discover the current or successfully closed initialization pin, using + /// exact indexed operation/lease and immutable outcome bindings. Historical metadata grants knowledge, not Write. + pub async fn load_initialization( + client: &CellClient, + target: &CellTarget, + store: &ArtifactStore, + input: &BeginRequest, + ) -> Result, RootRecoveryError> { + if crate::repository_target(target.tenant(), target.application(), input.repository)? + != *target + || store.repository() != input.repository + { + return Err(RootRecoveryError::Context); + } + let sql = SqlCell::::new(client.clone(), target.clone())?; + let observed = sql.query(None, statement( + "SELECT o.incarnation,o.admission_sequence FROM catalog_operations o JOIN catalog_leases l ON l.incarnation=o.incarnation AND l.admission_sequence=o.admission_sequence WHERE o.id=?1 AND o.actor=?2 AND o.request_digest=?3 AND o.generation=0 AND l.recovery IS NOT NULL UNION ALL SELECT l.incarnation,l.admission_sequence FROM catalog_initialization i JOIN catalog_leases l ON l.incarnation=i.incarnation AND l.admission_sequence=i.admission_sequence WHERE i.id=?1 AND i.actor=?2 AND i.request_digest=?3 AND l.generation=0 AND l.recovery IS NOT NULL", + vec![blob(input.operation),SqlValue::Text(input.actor.clone()),blob(input.request_digest)] + )).await.map_err(|error| RootRecoveryError::Query(Box::new(error)))?; + if rows(&observed.output)?.len() > 1 { + return Err(RootRecoveryError::Context); + } + let Some([incarnation, sequence]) = rows(&observed.output)?.first().map(Vec::as_slice) + else { + if rows(&observed.output)?.is_empty() { + return Ok(None); + } + return Err(RootRecoveryError::Context); + }; + let loaded = Self::load_pin( + client, + target, + store, + IncarnationId::from_bytes(fixed(incarnation)?), + unsigned(sequence)?, + None, + ) + .await? + .ok_or(RootRecoveryError::Context)?; + if loaded.record.kind != Kind::Initialization + || loaded.record.check.actor != input.actor + || loaded.token().operation != input.operation + || loaded.token().request_digest != input.request_digest + { + return Err(RootRecoveryError::Context); + } + Ok(Some(loaded)) + } + /// Original durable phase knowledge wins before body reads and fresh + /// custody. Only authoritative SDK absence can execute the saved bytes. + pub async fn recover_initialization( + &self, + client: &CellClient, + store: &ArtifactStore, + authority: &PreparationAuthority, + ) -> Result, PublicationError> { + self.dispatch_initialization(client, store, authority, None) + .await + } + pub(in crate::packs::publication) async fn dispatch_initialization( + &self, + client: &CellClient, + store: &ArtifactStore, + authority: &PreparationAuthority, + original: Option<&PreparationSession>, + ) -> Result, PublicationError> { + if self.record.kind != Kind::Initialization { + return Err(PublicationError::Recovery { + evidence: Box::new(self.evidence().clone()), + source: Box::new(RootRecoveryError::Context), + }); + } + let result = self + .dispatch_command::(client, store, authority, false, original) + .await + .map_err(|error| { + error.publication(self.evidence(), PublicationError::Initialization) + })?; + if matches!(result.output, InitializationReply::Denied(_)) { + return Err(PublicationError::Initialization(InvocationError::Rejected( + Box::new(result), + ))); + } + Ok(result) + } +} diff --git a/crates/canopy-server/src/packs/publication/recovery/mod.rs b/crates/canopy-server/src/packs/publication/recovery/mod.rs index 7a3c077a..918aeccb 100644 --- a/crates/canopy-server/src/packs/publication/recovery/mod.rs +++ b/crates/canopy-server/src/packs/publication/recovery/mod.rs @@ -16,6 +16,7 @@ use cellule_runtime::{ }; pub(in crate::packs::publication) mod archive; mod codec; +mod initialization; mod ready; mod supervisor; pub use archive::{ @@ -34,10 +35,12 @@ pub(in crate::packs::publication) use phase::execute; pub(in crate::packs::publication) use phase::normalize_root; const ROOT_BYTES: u32 = 8192; -const DOMAIN: &[u8] = b"canopy.publication-command-recovery.v3\0"; +const DOMAIN: &[u8] = b"canopy.publication-command-recovery.v4\0"; #[derive(Debug, thiserror::Error)] pub enum RootRecoveryError { + #[error("closed initialization graph failed")] + Initialization(#[from] super::initialization::InitializationVerificationError), #[error("closed native audit graph failed")] Audit(#[from] NativeResultError), #[error("terminal release command preparation failed")] @@ -74,10 +77,13 @@ pub(super) enum Kind { Publish, Outcome, Policy, + Initialization, } impl Kind { fn body_limit(self) -> u32 { - if self == Self::Policy { + if self == Self::Initialization { + INITIALIZATION_BYTES + } else if self == Self::Policy { REF_POLICY_PAGE_BYTES } else { ROOT_COMPLETION_BYTES @@ -305,27 +311,33 @@ impl RegisteredRootRecovery { &self, client: &CellClient, store: &ArtifactStore, + authority: &PreparationAuthority, ) -> Result, PublicationError> { - self.dispatch_root(client, store, None).await + self.dispatch_root(client, store, authority, None).await } async fn dispatch_root( &self, client: &CellClient, store: &ArtifactStore, + authority: &PreparationAuthority, original: Option<&PreparationSession>, ) -> Result, PublicationError> { let result = match self.record.kind { Kind::Publish => { - self.dispatch_command::(client, store, false, original) + self.dispatch_command::(client, store, authority, false, original) .await } Kind::Outcome => { - self.dispatch_command::(client, store, false, original) - .await + self.dispatch_command::( + client, store, authority, false, original, + ) + .await + } + Kind::Policy | Kind::Initialization => { + Err(AttemptError::Invocation(InvocationError::NotStarted( + Error::Command("recovery kind requires typed phase dispatch"), + ))) } - Kind::Policy => Err(AttemptError::Invocation(InvocationError::NotStarted( - Error::Command("policy recovery requires phase dispatch"), - ))), }; match result { Ok(value) => phase::normalize_root(Ok(value)).map_err(PublicationError::RootPush), @@ -336,11 +348,13 @@ impl RegisteredRootRecovery { &self, client: &CellClient, store: &ArtifactStore, + authority: &PreparationAuthority, refusing: &std::sync::atomic::AtomicBool, ) -> Result { self.dispatch_bound( client, store, + authority, refusing, None, #[cfg(test)] @@ -352,19 +366,29 @@ impl RegisteredRootRecovery { &self, client: &CellClient, store: &ArtifactStore, + authority: &PreparationAuthority, refusing: &std::sync::atomic::AtomicBool, original: Option<&PreparationSession>, #[cfg(test)] refusal_fault: Option<&std::sync::atomic::AtomicU8>, ) -> Result { + if self.record.kind == Kind::Initialization { + return self + .dispatch_initialization(client, store, authority, original) + .await + .map(PublicationOutcome::Initialization); + } if self.record.kind != Kind::Policy { let result = match original { - Some(original) => self.dispatch_root(client, store, Some(original)).await, - None => self.dispatch(client, store).await, + Some(original) => { + self.dispatch_root(client, store, authority, Some(original)) + .await + } + None => self.dispatch(client, store, authority).await, }; return result.map(PublicationOutcome::RootPush); } let result = self - .dispatch_command::(client, store, false, original) + .dispatch_command::(client, store, authority, false, original) .await; let refused = match &result { Ok(value) => { @@ -417,7 +441,7 @@ impl RegisteredRootRecovery { ))); } let outcome = self - .dispatch_command::(client, store, true, original) + .dispatch_command::(client, store, authority, true, original) .await; #[cfg(test)] if fault == 2 { @@ -475,6 +499,7 @@ impl RegisteredRootRecovery { &self, client: &CellClient, store: &ArtifactStore, + authority: &PreparationAuthority, refusal: bool, original: Option<&PreparationSession>, ) -> Result, AttemptError> { @@ -517,31 +542,39 @@ impl RegisteredRootRecovery { // would change both fencing semantics and refusal behavior: the final // transaction must still be able to select rejection after ACL loss. // Standalone recovery reacquires custody for positive work. - let session = if refusal_only || original.is_some() { - None - } else { - match PreparationSession::open( - client.clone(), - self.evidence().target().clone(), - self.record.check.clone(), - None, - ) - .await - { - Ok(session) => Some(session), - Err(error) => { - if let Some(known) = self.known::(client, store, refusal).await? { - return Ok(known); + // Initialization performs no new preparation or native work. Its + // frozen receiver checks Admin, the actual owner, live pin, checkpoint + // and pristine roots in the committing transaction. Requiring a fresh + // Write query here would hide an expired/revoked attempt before that + // original command could record its definitive denial. Bound live + // initialization still retains and checks its original local guard. + let session = + if refusal_only || self.record.kind == Kind::Initialization || original.is_some() { + None + } else { + match PreparationSession::open( + client.clone(), + self.evidence().target().clone(), + self.record.check.clone(), + None, + authority.clone(), + ) + .await + { + Ok(session) => Some(session), + Err(error) => { + if let Some(known) = self.known::(client, store, refusal).await? { + return Ok(known); + } + return Err(AttemptError::Invocation(InvocationError::NotStarted( + Error::Facility { + name: "publication recovery custody", + source: Box::new(error), + }, + ))); } - return Err(AttemptError::Invocation(InvocationError::NotStarted( - Error::Facility { - name: "publication recovery custody", - source: Box::new(error), - }, - ))); } - } - }; + }; // A command can settle while body I/O or custody acquisition is in flight. if let Some(known) = self.known::(client, store, refusal).await? { return Ok(known); diff --git a/crates/canopy-server/src/packs/publication/recovery/phase.rs b/crates/canopy-server/src/packs/publication/recovery/phase.rs index 2e897b68..90219713 100644 --- a/crates/canopy-server/src/packs/publication/recovery/phase.rs +++ b/crates/canopy-server/src/packs/publication/recovery/phase.rs @@ -84,7 +84,12 @@ pub(super) struct Journal { impl Journal { fn validate(&self, record: &Record) -> Result<(), CodecError> { if let Some(primary) = &self.primary { - let denied = if record.kind == Kind::Policy { + let denied = if record.kind == Kind::Initialization { + matches!( + primary.decode_reply::()?, + InitializationReply::Denied(_) + ) + } else if record.kind == Kind::Policy { matches!( primary.decode_reply::()?, RefPolicyReply::Denied(_) @@ -140,7 +145,12 @@ impl Journal { let Some(primary) = &self.primary else { return Ok(false); }; - Ok(if record.kind == Kind::Policy { + Ok(if record.kind == Kind::Initialization { + matches!( + primary.decode_reply::()?, + InitializationReply::Denied(_) + ) + } else if record.kind == Kind::Policy { matches!(primary.decode_reply::()?, RefPolicyReply::Registered(value) if value.valid) } else { matches!( @@ -340,7 +350,7 @@ impl RegisteredRootRecovery { let sql = SqlCell::::new(client.clone(), self.evidence().target().clone())?; let result = sql.query(None, SqlBatch { statements: vec![ - SqlStatement { sql: "SELECT recovery,recovery_phase FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 UNION ALL SELECT recovery,recovery_phase FROM pushes WHERE id=?3 AND recovery IS NOT NULL AND NOT EXISTS(SELECT 1 FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2)".into(), parameters: vec![blob(self.token().owner.incarnation.as_bytes()), number(self.token().attempt)?, blob(self.token().operation)] }, + SqlStatement { sql: "SELECT recovery,recovery_phase FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 UNION ALL SELECT recovery,recovery_phase FROM catalog_recovery_receipts WHERE incarnation=?1 AND admission_sequence=?2 AND operation=?3 AND NOT EXISTS(SELECT 1 FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2)".into(), parameters: vec![blob(self.token().owner.incarnation.as_bytes()), number(self.token().attempt)?, blob(self.token().operation)] }, SqlStatement { sql: "SELECT push_cert_seed FROM repository_identity WHERE singleton=1".into(), parameters: vec![] }, ] }).await.map_err(|error| RootRecoveryError::Query(Box::new(error)))?; let Some([SqlValue::Blob(bytes), saved]) = rows(&result.output)?.first().map(Vec::as_slice) diff --git a/crates/canopy-server/src/packs/publication/recovery/ready.rs b/crates/canopy-server/src/packs/publication/recovery/ready.rs index c98e9759..ce064874 100644 --- a/crates/canopy-server/src/packs/publication/recovery/ready.rs +++ b/crates/canopy-server/src/packs/publication/recovery/ready.rs @@ -8,6 +8,7 @@ pub struct ReadyRootRecovery { recovery: std::sync::Arc, client: CellClient, store: ArtifactStore, + authority: PreparationAuthority, refusing: std::sync::Arc, #[cfg(test)] refusal_fault: std::sync::Arc, @@ -20,9 +21,15 @@ impl RegisteredRootRecovery { self, client: CellClient, store: ArtifactStore, + authority: PreparationAuthority, ) -> Result { target_matches(self.evidence().target(), &store, &self.record.check)?; - Ok(ReadyRootRecovery::from_verified(self, client, store)) + if !authority.matches(self.evidence().target()) { + return Err(RootRecoveryError::Context); + } + Ok(ReadyRootRecovery::from_verified( + self, client, store, authority, + )) } } impl ReadyRootRecovery { @@ -41,11 +48,13 @@ impl ReadyRootRecovery { recovery: RegisteredRootRecovery, client: CellClient, store: ArtifactStore, + authority: PreparationAuthority, ) -> Self { Self { recovery: std::sync::Arc::new(recovery), client, store, + authority, refusing: std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)), #[cfg(test)] refusal_fault: std::sync::Arc::new(std::sync::atomic::AtomicU8::new(0)), @@ -92,6 +101,10 @@ impl ReadyRootRecovery { PublicationError::RootPush(InvocationError::Pending(Box::new( saved.snapshot.evidence().clone(), ))) + } else if self.recovery.record.kind == Kind::Initialization { + PublicationError::Initialization(InvocationError::Pending(Box::new( + self.recovery.evidence().clone(), + ))) } else if self.recovery.record.kind == Kind::Policy { PublicationError::PolicyPage(InvocationError::Pending(Box::new( self.recovery.evidence().clone(), @@ -122,6 +135,7 @@ impl ReadyRootRecovery { .dispatch_bound( &self.client, &self.store, + &self.authority, &self.refusing, Some(original), #[cfg(test)] @@ -131,7 +145,7 @@ impl ReadyRootRecovery { } None => { self.recovery - .dispatch_any(&self.client, &self.store, &self.refusing) + .dispatch_any(&self.client, &self.store, &self.authority, &self.refusing) .await } }; diff --git a/crates/canopy-server/src/packs/publication/recovery/registration.rs b/crates/canopy-server/src/packs/publication/recovery/registration.rs index 3920ff85..37a38a10 100644 --- a/crates/canopy-server/src/packs/publication/recovery/registration.rs +++ b/crates/canopy-server/src/packs/publication/recovery/registration.rs @@ -5,7 +5,7 @@ pub struct RegisterRootRecovery; impl Command for RegisterRootRecovery { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 39; - const CODEC_VERSION: u32 = 3; + const CODEC_VERSION: u32 = 4; type Input = RootRecoveryCertificate; type Output = RootRecoveryReply; fn execute( diff --git a/crates/canopy-server/src/packs/publication/recovery/supervisor.rs b/crates/canopy-server/src/packs/publication/recovery/supervisor.rs index 2b9656ac..b54027d0 100644 --- a/crates/canopy-server/src/packs/publication/recovery/supervisor.rs +++ b/crates/canopy-server/src/packs/publication/recovery/supervisor.rs @@ -1,6 +1,7 @@ //! Service-owned restart discovery over the existing independent attempt pins. //! No local outbox, new identity or caller-supplied actor is used. The original //! receiver still decides custody; discovery never adopts an old owner fence. +use super::super::scan::ScanControl; use super::*; use std::{sync::Arc, time::Duration}; use tokio::{sync::watch, task::JoinHandle}; @@ -24,7 +25,7 @@ impl Default for RecoveryScanLimits { } } impl RecoveryScanLimits { - fn validate(self) -> Result<(), RootRecoveryError> { + pub(in crate::packs::publication) fn validate(self) -> Result<(), RootRecoveryError> { if self.page == 0 || self.page > MAX_PAGE || self.interval < Duration::from_millis(10) @@ -68,7 +69,7 @@ impl RecoveryScanStats { /// retain its unresolved tickets and their original admission reservations. #[must_use] pub struct RecoverySupervisor { - stop: watch::Sender, + control: ScanControl, stats: watch::Receiver, task: Option>, } @@ -78,9 +79,18 @@ impl RecoverySupervisor { target: CellTarget, store: ArtifactStore, coordinator: PublicationCoordinator, - limits: RecoveryScanLimits, + settings: RecoveryScanSettings, + authority: PreparationAuthority, ) -> Result { - Self::start_inner(client, target, store, coordinator, limits, None) + Self::start_inner( + client, + target, + store, + coordinator, + settings, + authority, + None, + ) } /// The service supplies current repository administration and actual owner /// custody. Closed attempts release through the same fair maintenance queue; @@ -90,7 +100,8 @@ impl RecoverySupervisor { target: CellTarget, store: ArtifactStore, coordinator: PublicationCoordinator, - limits: RecoveryScanLimits, + settings: RecoveryScanSettings, + authority: PreparationAuthority, maintenance: MaintenanceRequest, ) -> Result { if maintenance.repository != store.repository() { @@ -102,7 +113,8 @@ impl RecoverySupervisor { target, store, coordinator, - limits, + settings, + authority, Some(maintenance), ) } @@ -111,49 +123,58 @@ impl RecoverySupervisor { target: CellTarget, store: ArtifactStore, coordinator: PublicationCoordinator, - limits: RecoveryScanLimits, + settings: RecoveryScanSettings, + authority: PreparationAuthority, maintenance: Option, ) -> Result { - limits.validate()?; - if !coordinator.matches_target(&target) + settings.validate()?; + if !authority.matches(&target) + || !coordinator.matches_target(&target) || crate::repository_target(target.tenant(), target.application(), store.repository())? != target { return Err(RootRecoveryError::Context); } let sql = SqlCell::::new(client.clone(), target.clone())?; - let (stop, stopping) = watch::channel(false); + let control = ScanControl::default(); let (updates, stats) = watch::channel(RecoveryScanStats::default()); - let task = tokio::spawn(run( + let task = settings.spawn(run( Scan { client, target, store, coordinator, + authority, maintenance, }, sql, - limits, - stopping, + settings.clone(), + control.clone(), updates, )); Ok(Self { - stop, + control, stats, task: Some(task), }) } + pub(crate) async fn pause(&self) { + self.control.pause().await; + } + pub(crate) fn resume(&self) { + self.control.resume(); + } pub fn stats(&self) -> RecoveryScanStats { self.stats.borrow().clone() } pub async fn shutdown(mut self) -> Result { - self.stop.send_replace(true); + self.control.stop(); self.task.take().expect("owned restart scanner").await } } impl Drop for RecoverySupervisor { fn drop(&mut self) { - self.stop.send_replace(true); + self.control.stop(); } } @@ -211,6 +232,7 @@ struct Scan { target: CellTarget, store: ArtifactStore, coordinator: PublicationCoordinator, + authority: PreparationAuthority, maintenance: Option, } impl Scan { @@ -260,6 +282,10 @@ impl Scan { if journal.terminal(®istered.record)?.is_some() && let Some(maintenance) = &self.maintenance { + if !registered.attempt_closed(&self.client).await? { + stats.deferred = stats.deferred.saturating_add(1); + return Ok(()); + } let identity = crate::server::mutation_identity().map_err(|source| Error::Facility { name: "terminal retirement identity", @@ -298,7 +324,11 @@ impl Scan { } match self .coordinator - .submit(registered.ready(self.client.clone(), self.store.clone())?) + .submit(registered.ready( + self.client.clone(), + self.store.clone(), + self.authority.clone(), + )?) .await { Ok(_) => stats.submitted = stats.submitted.saturating_add(1), @@ -324,15 +354,29 @@ impl Scan { async fn run( scan: Scan, sql: SqlCell, - limits: RecoveryScanLimits, - mut stopping: watch::Receiver, + settings: RecoveryScanSettings, + control: ScanControl, updates: watch::Sender, ) -> RecoveryScanStats { let mut stats = RecoveryScanStats::default(); let mut after = Cursor::default(); loop { - if *stopping.borrow() { + let Some(round) = control.enter(&settings).await else { return stats; + }; + let permit = match settings.acquire().await { + Ok(permit) => permit, + Err(_) => { + stats.deferred = stats.deferred.saturating_add(1); + updates.send_replace(stats.clone()); + drop(round); + settings.delay(&control).await; + continue; + } + }; + if control.interrupted(&settings) { + drop((permit, round)); + continue; } if scan.maintenance.is_some() { match scan.coordinator.recover_terminal_releases().await { @@ -348,14 +392,14 @@ async fn run( ), } } - match page(&sql, after, limits.page).await { + match page(&sql, after, settings.limits.page).await { Ok(keys) => { if keys.is_empty() { after = Cursor::default(); stats.passes = stats.passes.saturating_add(1); } for key in keys { - if *stopping.borrow() { + if control.interrupted(&settings) { break; } stats.scanned = stats.scanned.saturating_add(1); @@ -367,13 +411,11 @@ async fn run( } Err(error) => stats.failed(error), } + // Publish the completed round before releasing its quiescence guard: + // a successful pause also makes its diagnostics stable. updates.send_replace(stats.clone()); - tokio::select! { - _ = tokio::time::sleep(limits.interval) => {}, - changed = stopping.changed() => { - if changed.is_err() || *stopping.borrow() { return stats; } - } - } + drop((permit, round)); + settings.delay(&control).await; } } diff --git a/crates/canopy-server/src/packs/publication/registry.rs b/crates/canopy-server/src/packs/publication/registry.rs new file mode 100644 index 00000000..eaadd40c --- /dev/null +++ b/crates/canopy-server/src/packs/publication/registry.rs @@ -0,0 +1,140 @@ +//! One bounded operation contract shared by production and qualification. +use super::*; +use cellule_runtime::registry::OperationDescriptor; + +const fn command(input_limit: u32, output_limit: u32) -> OperationDescriptor { + OperationDescriptor { + id: C::ID, + codec_version: C::CODEC_VERSION, + schema_min: 1, + schema_max: 1, + input_limit, + output_limit, + } +} +const fn query(input_limit: u32, output_limit: u32) -> OperationDescriptor { + OperationDescriptor { + id: Q::ID, + codec_version: Q::CODEC_VERSION, + schema_min: 1, + schema_max: 1, + input_limit, + output_limit, + } +} + +pub(crate) const COMMANDS: [OperationDescriptor; 17] = [ + crate::operation(1), + command::(4096, 4096), + command::(4096, 4096), + command::(4096, 4096), + command::(4096, 4096), + command::(4096, 4096), + command::(INITIALIZATION_BYTES, 512), + command::(REF_POLICY_PAGE_BYTES, 128), + command::(4096, 128), + command::(ROOT_COMPLETION_BYTES, 512), + command::(ROOT_COMPLETION_BYTES, 512), + command::(4096, 4096), + command::(4096, 128), + command::(4096, 4096), + command::(1024, 512), + command::(1024, 128), + command::(1024, 128), +]; +pub(crate) const QUERIES: [OperationDescriptor; 11] = [ + crate::operation(2), + query::(4096, 4096), + query::(4096, 4096), + query::(4096, 4096), + query::(4096, 4096), + query::(4096, 4096), + query::(4096, 512), + query::(4096, 128), + query::(4096, 512), + query::(1024, 1024), + query::(1024, 512), +]; + +#[cfg(test)] +mod tests { + use super::*; + use crate::{CanopyApplication, RepositoryModule, build_descriptor}; + use cellule_app::CellApplication; + use cellule_runtime::CellModule; + + #[test] + fn production_registers_only_the_packed_command_contract() -> cellule_runtime::Result<()> { + let application = CanopyApplication::compile(build_descriptor( + include_bytes!("../../../../../Cargo.lock"), + env!("CARGO_PKG_VERSION"), + ))?; + assert!( + application + .registry() + .module_code(RepositoryModule::NAME) + .is_some() + ); + let descriptor = RepositoryModule.descriptor(); + let ids: Vec<_> = descriptor + .commands + .iter() + .map(|operation| operation.id) + .collect(); + assert_eq!( + ids, + vec![ + 1, 14, 16, 17, 22, 29, 31, 33, 35, 36, 38, 39, 40, 41, 42, 43, 46 + ] + ); + assert_eq!( + descriptor + .queries + .iter() + .map(|operation| operation.id) + .collect::>(), + vec![2, 15, 21, 23, 27, 30, 32, 34, 37, 47, 48] + ); + for (id, codec, input, output) in [ + ( + 33, + RegisterRefPolicyPage::CODEC_VERSION, + REF_POLICY_PAGE_BYTES, + 128, + ), + ( + 36, + CompleteRootPush::CODEC_VERSION, + ROOT_COMPLETION_BYTES, + 512, + ), + ( + 38, + CompleteRootOutcome::CODEC_VERSION, + ROOT_COMPLETION_BYTES, + 512, + ), + (39, RegisterRootRecovery::CODEC_VERSION, 4096, 4096), + (40, ReleaseTerminalRecovery::CODEC_VERSION, 4096, 128), + (41, RegisterCustodyIntent::CODEC_VERSION, 4096, 4096), + (42, ExecuteCustody::CODEC_VERSION, 1024, 512), + (43, StopCustodyIntent::CODEC_VERSION, 1024, 128), + (46, ReleaseServingPin::CODEC_VERSION, 1024, 128), + ] { + let operation = descriptor + .commands + .iter() + .find(|value| value.id == id) + .unwrap(); + assert_eq!( + ( + operation.codec_version, + operation.input_limit, + operation.output_limit + ), + (codec, input, output) + ); + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/publication/scan.rs b/crates/canopy-server/src/packs/publication/scan.rs new file mode 100644 index 00000000..7b47b19c --- /dev/null +++ b/crates/canopy-server/src/packs/publication/scan.rs @@ -0,0 +1,192 @@ +//! Shared read-round admission and quiescence for repository recovery owners. +use super::{RecoveryScanLimits, RootRecoveryError}; +use crate::{AdmissionPermit, ReadIdentity, admission::AccountAdmission}; +use std::sync::{Arc, Mutex}; +use tokio::sync::Notify; +use tokio_util::{sync::CancellationToken, task::TaskTracker}; + +#[derive(Clone)] +pub struct RecoveryScanBudget { + inner: Arc, +} +struct Budget { + limit: usize, + admission: AccountAdmission, + stop: CancellationToken, + tasks: TaskTracker, +} +impl RecoveryScanBudget { + pub fn new(limit: u16, tasks: TaskTracker) -> Result { + if !(2..=64).contains(&limit) { + return Err(RootRecoveryError::InvalidScanLimits); + } + Ok(Self { + inner: Arc::new(Budget { + limit: usize::from(limit), + admission: AccountAdmission::new( + usize::from(limit), + "node recovery reads", + "account recovery reads", + ), + stop: CancellationToken::new(), + tasks, + }), + }) + } + /// Account comes from trusted repository administration, not a viewer label. + pub fn settings(&self, limits: RecoveryScanLimits, account: &str) -> RecoveryScanSettings { + RecoveryScanSettings { + limits, + budget: self.clone(), + account: Arc::from(account), + } + } + /// Stop discovery between owned reads; never cancel a current round. + pub fn close(&self) { + self.inner.stop.cancel(); + } + pub fn in_flight(&self) -> usize { + self.inner.limit - self.inner.admission.available() + } +} + +#[derive(Clone)] +pub struct RecoveryScanSettings { + pub limits: RecoveryScanLimits, + budget: RecoveryScanBudget, + account: Arc, +} +impl RecoveryScanSettings { + pub(super) fn validate(&self) -> Result<(), RootRecoveryError> { + self.limits.validate()?; + if self.account.is_empty() || self.account.len() > 4096 { + return Err(RootRecoveryError::InvalidScanLimits); + } + Ok(()) + } + pub(super) async fn acquire(&self) -> Result { + self.budget + .inner + .admission + .acquire(ReadIdentity::Account(&self.account)) + .await + } + pub(super) fn spawn(&self, future: F) -> tokio::task::JoinHandle + where + F: std::future::Future + Send + 'static, + F::Output: Send + 'static, + { + self.budget.inner.tasks.spawn(future) + } + pub(super) async fn delay(&self, control: &ScanControl) { + let changed = control.inner.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if control.interrupted(self) { + return; + } + tokio::select! { + _ = tokio::time::sleep(self.limits.interval) => {}, + _ = changed => {}, + _ = self.budget.inner.stop.cancelled() => {}, + } + } +} + +#[derive(Clone, Default)] +pub(super) struct ScanControl { + inner: Arc, +} +#[derive(Default)] +struct Control { + state: Mutex, + changed: Notify, +} +#[derive(Default)] +struct ControlState { + paused: bool, + stopped: bool, + active: bool, +} +impl ScanControl { + pub(super) async fn enter(&self, settings: &RecoveryScanSettings) -> Option { + loop { + let changed = self.inner.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + { + let mut state = self.inner.state.lock().expect("recovery scan control"); + if state.stopped || settings.budget.inner.stop.is_cancelled() { + return None; + } + if !state.paused { + assert!(!state.active, "one owner per recovery scanner"); + state.active = true; + return Some(ActiveScan(self.clone())); + } + } + tokio::select! { + _ = changed => {}, + _ = settings.budget.inner.stop.cancelled() => return None, + } + } + } + pub(super) fn interrupted(&self, settings: &RecoveryScanSettings) -> bool { + let state = self.inner.state.lock().expect("recovery scan control"); + state.paused || state.stopped || settings.budget.inner.stop.is_cancelled() + } + pub(super) async fn pause(&self) { + self.inner + .state + .lock() + .expect("recovery scan control") + .paused = true; + self.inner.changed.notify_waiters(); + loop { + let changed = self.inner.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if !self + .inner + .state + .lock() + .expect("recovery scan control") + .active + { + return; + } + changed.await; + } + } + pub(super) fn resume(&self) { + self.inner + .state + .lock() + .expect("recovery scan control") + .paused = false; + self.inner.changed.notify_waiters(); + } + pub(super) fn stop(&self) { + self.inner + .state + .lock() + .expect("recovery scan control") + .stopped = true; + self.inner.changed.notify_waiters(); + } +} +pub(super) struct ActiveScan(ScanControl); +impl Drop for ActiveScan { + fn drop(&mut self) { + self.0 + .inner + .state + .lock() + .expect("recovery scan control") + .active = false; + self.0.inner.changed.notify_waiters(); + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/publication/scan/tests.rs b/crates/canopy-server/src/packs/publication/scan/tests.rs new file mode 100644 index 00000000..90855bd1 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/scan/tests.rs @@ -0,0 +1,119 @@ +use super::*; +use tokio::time::{Duration, timeout}; + +#[tokio::test] +async fn scan_pause_joins_active_work_and_preserves_resume_and_terminal_stop() { + let tasks = TaskTracker::new(); + let budget = RecoveryScanBudget::new(4, tasks).unwrap(); + let settings = budget.settings(RecoveryScanLimits::default(), "owner"); + let control = ScanControl::default(); + let round = control.enter(&settings).await.unwrap(); + let pause = control.pause(); + tokio::pin!(pause); + assert!( + timeout(Duration::from_millis(20), &mut pause) + .await + .is_err() + ); + drop(round); + timeout(Duration::from_secs(1), &mut pause).await.unwrap(); + assert!( + timeout(Duration::from_millis(20), control.enter(&settings)) + .await + .is_err() + ); + control.resume(); + let round = control.enter(&settings).await.unwrap(); + control.stop(); + let pause = control.pause(); + tokio::pin!(pause); + assert!( + timeout(Duration::from_millis(20), &mut pause) + .await + .is_err() + ); + drop(round); + timeout(Duration::from_secs(1), &mut pause).await.unwrap(); + control.resume(); + assert!(control.enter(&settings).await.is_none()); + // A stop observed before interval registration must not wait one interval. + timeout(Duration::from_millis(50), settings.delay(&control)) + .await + .unwrap(); +} + +#[tokio::test] +async fn shared_read_budget_bounds_account_rounds_and_tracks_shutdown_without_cancellation() { + let tasks = TaskTracker::new(); + let budget = RecoveryScanBudget::new(4, tasks.clone()).unwrap(); + let a = budget.settings(RecoveryScanLimits::default(), "a"); + let another_repository = budget.settings(RecoveryScanLimits::default(), "a"); + let b = budget.settings(RecoveryScanLimits::default(), "b"); + let p1 = a.acquire().await.unwrap(); + let p2 = another_repository.acquire().await.unwrap(); + assert!(another_repository.acquire().await.is_err()); + let p3 = b.acquire().await.unwrap(); + let p4 = b.acquire().await.unwrap(); + assert_eq!(budget.in_flight(), 4); + drop((p2, p3, p4)); + let (release, waiting) = tokio::sync::oneshot::channel(); + let (entered, observing) = tokio::sync::oneshot::channel(); + let scope = a.clone(); + let task = a.spawn(async move { + let control = ScanControl::default(); + let round = control.enter(&scope).await.unwrap(); + let _ = entered.send(()); + let _ = waiting.await; + drop((round, p1)); + assert!(control.enter(&scope).await.is_none()); + }); + observing.await.unwrap(); + budget.close(); + tasks.close(); + assert!( + timeout(Duration::from_millis(20), tasks.wait()) + .await + .is_err() + ); + assert_eq!(budget.in_flight(), 1); + release.send(()).unwrap(); + timeout(Duration::from_secs(1), tasks.wait()).await.unwrap(); + task.await.unwrap(); + assert_eq!(budget.in_flight(), 0); + assert!(ScanControl::default().enter(&a).await.is_none()); +} + +#[test] +fn scanner_settings_reject_invalid_rounds_and_account_keys() { + for limit in [0, 1, 65, u16::MAX] { + assert!(matches!( + RecoveryScanBudget::new(limit, TaskTracker::new()), + Err(RootRecoveryError::InvalidScanLimits) + )); + } + let budget = RecoveryScanBudget::new(2, TaskTracker::new()).unwrap(); + assert!( + budget + .settings(RecoveryScanLimits::default(), "") + .validate() + .is_err() + ); + assert!( + budget + .settings(RecoveryScanLimits::default(), &"a".repeat(4097)) + .validate() + .is_err() + ); + assert!( + budget + .settings( + RecoveryScanLimits { + page: 0, + ..RecoveryScanLimits::default() + }, + "a" + ) + .validate() + .is_err() + ); +} diff --git a/crates/canopy-server/src/packs/publication/schema.sql b/crates/canopy-server/src/packs/publication/schema.sql index 603f714a..6bf84189 100644 --- a/crates/canopy-server/src/packs/publication/schema.sql +++ b/crates/canopy-server/src/packs/publication/schema.sql @@ -65,14 +65,10 @@ CREATE TABLE pushes ( actor TEXT NOT NULL, request_digest BLOB NOT NULL CHECK(length(request_digest) = 32), initial_staging BLOB CHECK(initial_staging IS NULL OR (typeof(initial_staging)='blob' AND length(initial_staging) BETWEEN 1 AND 1024)), + initial_preparation BLOB CHECK(initial_preparation IS NULL OR (typeof(initial_preparation)='blob' AND length(initial_preparation) BETWEEN 1 AND 1024)), options TEXT NOT NULL DEFAULT '[]' CHECK(length(CAST(options AS BLOB)) <= 65536), response_id BLOB CHECK(response_id IS NULL OR length(response_id) = 16), completion_digest BLOB CHECK(completion_digest IS NULL OR length(completion_digest) = 32), - -- Closed attempts keep their original recovery headers and phase here, - -- independently of a preparation floor or creating-input lease. - recovery BLOB CHECK(recovery IS NULL OR (typeof(recovery)='blob' AND length(recovery) BETWEEN 1 AND 1024)), - recovery_phase BLOB CHECK(recovery_phase IS NULL OR (typeof(recovery_phase)='blob' AND length(recovery_phase) BETWEEN 1 AND 2048)), - recovery_release BLOB CHECK(recovery_release IS NULL OR (typeof(recovery_release)='blob' AND length(recovery_release) BETWEEN 1 AND 1024)), response_root BLOB CHECK(response_root IS NULL OR (typeof(response_root)='blob' AND length(response_root) BETWEEN 1 AND 128)), rejected INTEGER CHECK(rejected IN (0, 1)), rejection_reason TEXT, @@ -84,9 +80,6 @@ CREATE TABLE pushes ( CHECK((response_id IS NULL) = (rejected IS NULL)), CHECK((response_id IS NULL) = (completion_digest IS NULL)), CHECK(response_root IS NULL OR response_id IS NOT NULL), - CHECK((recovery IS NULL) = (recovery_phase IS NULL)), - CHECK((recovery IS NULL) = (recovery_release IS NULL)), - CHECK(recovery IS NULL OR response_root IS NOT NULL), CHECK(rejected IS NOT 1 OR publication IS NULL) ) WITHOUT ROWID; CREATE TRIGGER push_publication_immutable BEFORE UPDATE OF publication,publication_plan_digest ON pushes @@ -110,10 +103,6 @@ CREATE TRIGGER push_root_completion_retained BEFORE DELETE ON pushes WHEN OLD.response_root IS NOT NULL BEGIN SELECT RAISE(ABORT, 'root completion must be retained'); END; -CREATE TRIGGER push_recovery_archive_immutable BEFORE UPDATE OF recovery,recovery_phase,recovery_release ON pushes -WHEN OLD.recovery IS NOT NULL AND (NEW.recovery IS NOT OLD.recovery OR NEW.recovery_phase IS NOT OLD.recovery_phase OR NEW.recovery_release IS NOT OLD.recovery_release) -BEGIN SELECT RAISE(ABORT, 'closed recovery is immutable'); END; - CREATE TRIGGER push_initial_staging_immutable BEFORE UPDATE OF initial_staging ON pushes WHEN OLD.initial_staging IS NOT NULL AND NEW.initial_staging IS NOT OLD.initial_staging BEGIN SELECT RAISE(ABORT, 'initial staging receipt is immutable'); END; @@ -121,6 +110,13 @@ CREATE TRIGGER push_initial_staging_retained BEFORE DELETE ON pushes WHEN OLD.initial_staging IS NOT NULL BEGIN SELECT RAISE(ABORT, 'initial staging receipt must be retained'); END; +CREATE TRIGGER push_initial_preparation_immutable BEFORE UPDATE OF initial_preparation ON pushes +WHEN OLD.initial_preparation IS NOT NULL AND NEW.initial_preparation IS NOT OLD.initial_preparation +BEGIN SELECT RAISE(ABORT, 'initial preparation receipt is immutable'); END; +CREATE TRIGGER push_initial_preparation_retained BEFORE DELETE ON pushes +WHEN OLD.initial_preparation IS NOT NULL +BEGIN SELECT RAISE(ABORT, 'initial preparation receipt must be retained'); END; + CREATE TABLE push_certificates ( digest BLOB PRIMARY KEY CHECK(length(digest) = 32), push_id BLOB NOT NULL UNIQUE REFERENCES pushes(id), @@ -365,6 +361,8 @@ INSERT INTO catalog_state VALUES(1, 0); -- fact retains the small empty catalog/ref metadata for exact logical recovery. CREATE TABLE catalog_initialization ( singleton INTEGER PRIMARY KEY CHECK(singleton=1), + incarnation BLOB NOT NULL CHECK(typeof(incarnation)='blob' AND length(incarnation)=16), + admission_sequence INTEGER NOT NULL CHECK(typeof(admission_sequence)='integer' AND admission_sequence>0), id BLOB NOT NULL UNIQUE CHECK(length(id)=16), actor TEXT NOT NULL, request_digest BLOB NOT NULL CHECK(length(request_digest)=32), @@ -394,6 +392,27 @@ CREATE TRIGGER catalog_compactions_not_replaced BEFORE INSERT ON catalog_compact WHEN EXISTS(SELECT 1 FROM catalog_compactions WHERE id=NEW.id) BEGIN SELECT RAISE(ABORT, 'compaction outcomes cannot be replaced'); END; +-- Closed attempts retain the same bounded authenticated header, phase and +-- original release receipt, independent of generation floors and pin quota. +-- Multiple attempts of one logical initialization have independent identities. +CREATE TABLE catalog_recovery_receipts ( + incarnation BLOB NOT NULL CHECK(typeof(incarnation)='blob' AND length(incarnation)=16), + admission_sequence INTEGER NOT NULL CHECK(typeof(admission_sequence)='integer' AND admission_sequence>0), + operation BLOB NOT NULL CHECK(typeof(operation)='blob' AND length(operation)=16), + recovery BLOB NOT NULL CHECK(typeof(recovery)='blob' AND length(recovery) BETWEEN 1 AND 1024), + recovery_phase BLOB NOT NULL CHECK(typeof(recovery_phase)='blob' AND length(recovery_phase) BETWEEN 1 AND 2048), + recovery_release BLOB NOT NULL CHECK(typeof(recovery_release)='blob' AND length(recovery_release) BETWEEN 1 AND 1024), + PRIMARY KEY(incarnation,admission_sequence) +) WITHOUT ROWID; +CREATE INDEX catalog_recovery_receipts_by_operation ON catalog_recovery_receipts(operation,incarnation,admission_sequence); +CREATE TRIGGER catalog_recovery_receipt_immutable BEFORE UPDATE ON catalog_recovery_receipts +BEGIN SELECT RAISE(ABORT, 'closed recovery is immutable'); END; +CREATE TRIGGER catalog_recovery_receipt_not_replaced BEFORE INSERT ON catalog_recovery_receipts +WHEN EXISTS(SELECT 1 FROM catalog_recovery_receipts WHERE incarnation=NEW.incarnation AND admission_sequence=NEW.admission_sequence) +BEGIN SELECT RAISE(ABORT, 'closed recovery cannot be replaced'); END; +CREATE TRIGGER catalog_recovery_receipt_retained BEFORE DELETE ON catalog_recovery_receipts +BEGIN SELECT RAISE(ABORT, 'closed recovery must be retained'); END; + -- Staging attempts retain their creating namespace with a NULL generation. -- A one-way late bind pins a generation floor and every later generation. This permits -- read-only frontier refresh without a new durable pin/Claim per publication. @@ -434,6 +453,7 @@ CREATE INDEX catalog_leases_by_expiry ON catalog_leases(expires_at_ms, incarnati CREATE INDEX catalog_leases_by_generation ON catalog_leases(generation, expires_at_ms); CREATE TRIGGER catalog_generations_retained BEFORE DELETE ON catalog_generations WHEN OLD.generation=0 OR OLD.generation >= (SELECT min(generation) FROM catalog_leases) + OR EXISTS(SELECT 1 FROM catalog_serving_pins WHERE generation=OLD.generation) BEGIN SELECT RAISE(ABORT, 'catalog generation is retained'); END; CREATE UNIQUE INDEX catalog_leases_by_artifact ON catalog_leases(artifact_operation); CREATE TRIGGER catalog_lease_not_replaced BEFORE INSERT ON catalog_leases @@ -471,9 +491,9 @@ BEGIN SELECT RAISE(ABORT, 'publication recovery requires exact phase append'); E -- Typed recovery/backup traversal must authorize releasing these pins. CREATE TRIGGER catalog_lease_recovery_retained BEFORE DELETE ON catalog_leases WHEN OLD.recovery IS NOT NULL AND NOT EXISTS( - SELECT 1 FROM pushes p WHERE p.id=OLD.operation AND p.response_root IS NOT NULL - AND p.recovery IS OLD.recovery AND p.recovery_phase IS OLD.recovery_phase - AND p.recovery_release IS NOT NULL) + SELECT 1 FROM catalog_recovery_receipts p WHERE p.incarnation=OLD.incarnation + AND p.admission_sequence=OLD.admission_sequence AND p.operation=OLD.operation + AND p.recovery IS OLD.recovery AND p.recovery_phase IS OLD.recovery_phase) BEGIN SELECT RAISE(ABORT, 'root recovery command is retained'); END; CREATE TABLE catalog_operations ( @@ -526,3 +546,44 @@ BEGIN SELECT RAISE(ABORT, 'push outcome bytes are immutable'); END; CREATE TRIGGER push_certificate_chunks_not_replaced BEFORE INSERT ON push_certificate_chunks WHEN EXISTS(SELECT 1 FROM push_certificate_chunks WHERE push_id=NEW.push_id AND part=NEW.part) BEGIN SELECT RAISE(ABORT, 'push outcome bytes cannot be replaced'); END; + +-- Bounded exact-command metadata exists before any upload namespace is granted. +-- One unresolved head per logical request. Rows represent custody transitions, +-- never Git objects, and historical grants are not generation retention roots. +CREATE TABLE catalog_custody_commands ( + purpose INTEGER NOT NULL CHECK(typeof(purpose)='integer' AND purpose IN (0,1)), + operation BLOB NOT NULL CHECK(typeof(operation)='blob' AND length(operation)=16), + step INTEGER NOT NULL CHECK(typeof(step)='integer' AND step BETWEEN 0 AND 65535), + incarnation BLOB NOT NULL CHECK(typeof(incarnation)='blob' AND length(incarnation)=16), + request_id BLOB NOT NULL CHECK(typeof(request_id)='blob' AND length(request_id)=16), + intent BLOB NOT NULL CHECK(typeof(intent)='blob' AND length(intent) BETWEEN 1 AND 4096), + phase BLOB CHECK(phase IS NULL OR (typeof(phase)='blob' AND length(phase) BETWEEN 1 AND 1024)), + stopped BLOB CHECK(stopped IS NULL OR (typeof(stopped)='blob' AND length(stopped) BETWEEN 1 AND 1024)), + granted_incarnation BLOB CHECK(granted_incarnation IS NULL OR (typeof(granted_incarnation)='blob' AND length(granted_incarnation)=16)), + granted_attempt INTEGER CHECK(granted_attempt IS NULL OR (typeof(granted_attempt)='integer' AND granted_attempt>0)), + CHECK(phase IS NULL OR stopped IS NULL), + CHECK(stopped IS NULL OR granted_attempt IS NULL), + CHECK((granted_incarnation IS NULL)=(granted_attempt IS NULL)), + CHECK(phase IS NOT NULL OR granted_attempt IS NULL), + PRIMARY KEY(purpose,operation,step), + UNIQUE(incarnation,request_id) +) WITHOUT ROWID; +CREATE INDEX catalog_custody_grants ON catalog_custody_commands(purpose,operation,granted_incarnation,granted_attempt,step DESC) WHERE granted_attempt IS NOT NULL; +CREATE UNIQUE INDEX catalog_custody_pending ON catalog_custody_commands(purpose,operation) WHERE phase IS NULL AND stopped IS NULL; +CREATE TRIGGER catalog_custody_identity_immutable BEFORE UPDATE OF purpose,operation,step,incarnation,request_id,intent ON catalog_custody_commands +WHEN NEW.purpose IS NOT OLD.purpose OR NEW.operation IS NOT OLD.operation OR NEW.step IS NOT OLD.step + OR NEW.incarnation IS NOT OLD.incarnation OR NEW.request_id IS NOT OLD.request_id OR NEW.intent IS NOT OLD.intent +BEGIN SELECT RAISE(ABORT, 'custody command identity is immutable'); END; +CREATE TRIGGER catalog_custody_phase_immutable BEFORE UPDATE OF phase,granted_incarnation,granted_attempt ON catalog_custody_commands +WHEN OLD.phase IS NOT NULL AND (NEW.phase IS NOT OLD.phase OR NEW.granted_incarnation IS NOT OLD.granted_incarnation OR NEW.granted_attempt IS NOT OLD.granted_attempt) +BEGIN SELECT RAISE(ABORT, 'custody command result is immutable'); END; +CREATE TRIGGER catalog_custody_not_replaced BEFORE INSERT ON catalog_custody_commands +WHEN EXISTS(SELECT 1 FROM catalog_custody_commands WHERE purpose=NEW.purpose AND operation=NEW.operation AND step=NEW.step) + OR EXISTS(SELECT 1 FROM catalog_custody_commands WHERE incarnation=NEW.incarnation AND request_id=NEW.request_id) +BEGIN SELECT RAISE(ABORT, 'custody command cannot be replaced'); END; +CREATE TRIGGER catalog_custody_retained BEFORE DELETE ON catalog_custody_commands +BEGIN SELECT RAISE(ABORT, 'custody command must be retained'); END; + +CREATE TRIGGER catalog_custody_stop_immutable BEFORE UPDATE OF stopped ON catalog_custody_commands +WHEN OLD.stopped IS NOT NULL AND NEW.stopped IS NOT OLD.stopped +BEGIN SELECT RAISE(ABORT, 'custody retirement is immutable'); END; diff --git a/crates/canopy-server/src/packs/publication/serving.rs b/crates/canopy-server/src/packs/publication/serving.rs new file mode 100644 index 00000000..0a524ae9 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving.rs @@ -0,0 +1,97 @@ +//! Certified read generations retained until owned workers have actually drained. +use super::*; +use cellule_runtime::{CellClient, CellTarget, MutationIdentity, PreparedCommand}; +use std::sync::Arc; +mod codec; +mod command_owner; +mod commands; +pub use command_owner::ReadyServingCommand; +mod lifecycle; +mod ownership; +mod pool; +pub use lifecycle::{ + ServingDrainObserver, ServingOwner, ServingOwnerError, ServingOwnerPhase, ServingOwnerStats, + ServingSnapshot, +}; +pub use ownership::MAX_SERVING_OWNERS; +pub use pool::{MAX_SERVING_GENERATIONS, ServingPool, ServingPoolLimits}; +mod session; +pub use commands::{ + AcquireServingPin, CheckServingPin, ReleaseServingPin, RenewServingPin, SelectServingGeneration, +}; +pub use session::{ + MAX_EDGE_PARENTS, NativeWorkspace, ReadyServingRelease, ResolvedServingRef, ServingContext, + ServingEdgePage, ServingPin, ServingReadBudget, ServingReadError, WorkspaceLimits, + WorkspaceStats, +}; + +pub const MAX_SERVING_PINS: u64 = 4096; +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ServingToken { + pub repository: [u8; 16], + pub reader: [u8; 16], + pub owner: OwnerFence, + pub admission_sequence: u64, + pub generation: u64, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AcquireServingRequest { + pub repository: [u8; 16], + pub reader: [u8; 16], + pub actor: Option, + pub lease_ms: u64, +} +/// Observe a current joint root. This neither retains it nor grants artifact I/O. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ServingSelection { + pub repository: [u8; 16], + pub actor: Option, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ServingCheck { + pub token: ServingToken, + pub actor: Option, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RenewServingRequest { + pub check: ServingCheck, + pub lease_ms: u64, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ServingLease { + pub token: ServingToken, + pub fact: GenerationFact, + pub format: ObjectFormat, + pub observed_at_ms: i64, + pub expires_at_ms: i64, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ServingDenial { + Unauthorized, + Conflict, + Uninitialized, + Stale, + Expired, + Capacity, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum ServingReply { + Granted(Box), + Denied(ServingDenial), +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum ServingReleaseReply { + Released, + Denied(ServingDenial), +} +/// Transport is untrusted until its purpose-separated MAC is checked. Only a +/// private service owner whose workers drained can issue this certificate. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ServingDrainProof(super::certificate::CertificateEnvelope); +#[derive(Clone, Debug, PartialEq, Eq)] +struct DrainData { + tenant: [u8; 16], + application: [u8; 16], + token: ServingToken, + administrator: String, +} diff --git a/crates/canopy-server/src/packs/publication/serving/codec.rs b/crates/canopy-server/src/packs/publication/serving/codec.rs new file mode 100644 index 00000000..848ae8c8 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/codec.rs @@ -0,0 +1,391 @@ +use super::*; +use crate::packs::directory::index::codec::fixed; +const DOMAIN: &[u8] = b"canopy.serving-workers-drained.v1\0"; +fn invalid() -> CodecError { + CodecError::Invalid("invalid serving pin") +} +fn actor(value: &Option) -> Result<(), CodecError> { + if let Some(value) = value { + validate_component(value).map_err(|_| invalid())?; + } + Ok(()) +} +fn duration(value: u64) -> Result<(), CodecError> { + if value == 0 || value > MAX_LEASE_MS { + return Err(invalid()); + } + Ok(()) +} +impl ServingToken { + pub(in crate::packs::publication) fn validate(&self) -> Result<(), CodecError> { + crate::validate_repository_id(self.repository).map_err(|_| invalid())?; + if self.reader == [0; 16] + || self.owner.epoch == 0 + || self.admission_sequence == 0 + || self.admission_sequence > i64::MAX as u64 + || self.generation == 0 + || self.generation > i64::MAX as u64 + { + return Err(invalid()); + } + Ok(()) + } +} +impl WireValue for ServingToken { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(&self.repository)?; + e.write_bytes(&self.reader)?; + e.write_bytes(self.owner.incarnation.as_bytes())?; + e.write_u64(self.owner.epoch)?; + e.write_u64(self.admission_sequence)?; + e.write_u64(self.generation) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + repository: fixed(d)?, + reader: fixed(d)?, + owner: OwnerFence { + incarnation: IncarnationId::from_bytes(fixed(d)?), + epoch: d.read_u64()?, + }, + admission_sequence: d.read_u64()?, + generation: d.read_u64()?, + }; + value.validate()?; + Ok(value) + } +} +impl WireValue for AcquireServingRequest { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + crate::validate_repository_id(self.repository).map_err(|_| invalid())?; + if self.reader == [0; 16] { + return Err(invalid()); + } + actor(&self.actor)?; + duration(self.lease_ms)?; + e.write_bytes(&self.repository)?; + e.write_bytes(&self.reader)?; + self.actor.encode(e)?; + e.write_u64(self.lease_ms) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + repository: fixed(d)?, + reader: fixed(d)?, + actor: Option::::decode(d)?, + lease_ms: d.read_u64()?, + }; + value.encode(&mut BoundedEncoder::new(1024)?)?; + Ok(value) + } +} +impl WireValue for ServingSelection { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + crate::validate_repository_id(self.repository).map_err(|_| invalid())?; + actor(&self.actor)?; + e.write_bytes(&self.repository)?; + self.actor.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + repository: fixed(d)?, + actor: Option::::decode(d)?, + }; + value.encode(&mut BoundedEncoder::new(1024)?)?; + Ok(value) + } +} +impl WireValue for ServingCheck { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + actor(&self.actor)?; + self.token.encode(e)?; + self.actor.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + token: ServingToken::decode(d)?, + actor: Option::::decode(d)?, + }; + actor(&value.actor)?; + Ok(value) + } +} +impl WireValue for RenewServingRequest { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + duration(self.lease_ms)?; + self.check.encode(e)?; + e.write_u64(self.lease_ms) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + check: ServingCheck::decode(d)?, + lease_ms: d.read_u64()?, + }; + duration(value.lease_ms)?; + Ok(value) + } +} +impl ServingLease { + pub(in crate::packs::publication) fn validate(&self) -> Result<(), CodecError> { + self.token.validate()?; + self.fact.validate()?; + if self.fact.generation != self.token.generation + || self.fact.refs.is_none() + || self.fact.catalog.is_none_or(|catalog| { + catalog.repository != self.token.repository || catalog.format != self.format + }) + || self.observed_at_ms < 0 + || self.expires_at_ms <= self.observed_at_ms + { + return Err(invalid()); + } + Ok(()) + } +} +impl WireValue for ServingLease { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + self.token.encode(e)?; + self.fact.encode(e)?; + e.write_u8(self.format.bytes() as u8)?; + e.write_i64(self.observed_at_ms)?; + e.write_i64(self.expires_at_ms) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + token: ServingToken::decode(d)?, + fact: GenerationFact::decode(d)?, + format: match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(invalid()), + }, + observed_at_ms: d.read_i64()?, + expires_at_ms: d.read_i64()?, + }; + value.validate()?; + Ok(value) + } +} +impl WireValue for ServingDenial { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + e.write_u8(match self { + Self::Unauthorized => 0, + Self::Conflict => 1, + Self::Uninitialized => 2, + Self::Stale => 3, + Self::Expired => 4, + Self::Capacity => 5, + }) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + Ok(match d.read_u8()? { + 0 => Self::Unauthorized, + 1 => Self::Conflict, + 2 => Self::Uninitialized, + 3 => Self::Stale, + 4 => Self::Expired, + 5 => Self::Capacity, + _ => return Err(invalid()), + }) + } +} +impl WireValue for ServingReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Granted(value) => { + e.write_u8(0)?; + value.encode(e) + } + Self::Denied(value) => { + e.write_u8(1)?; + value.encode(e) + } + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 0 => Ok(Self::Granted(Box::new(ServingLease::decode(d)?))), + 1 => Ok(Self::Denied(ServingDenial::decode(d)?)), + _ => Err(invalid()), + } + } +} +impl WireValue for ServingReleaseReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Released => e.write_u8(0), + Self::Denied(value) => { + e.write_u8(1)?; + value.encode(e) + } + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 0 => Ok(Self::Released), + 1 => Ok(Self::Denied(ServingDenial::decode(d)?)), + _ => Err(invalid()), + } + } +} +impl WireValue for DrainData { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + validate_component(&self.administrator).map_err(|_| invalid())?; + e.write_bytes(DOMAIN)?; + e.write_bytes(&self.tenant)?; + e.write_bytes(&self.application)?; + self.token.encode(e)?; + e.write_text(&self.administrator) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(invalid()); + } + let value = Self { + tenant: fixed(d)?, + application: fixed(d)?, + token: ServingToken::decode(d)?, + administrator: d.read_text()?.into(), + }; + validate_component(&value.administrator).map_err(|_| invalid())?; + Ok(value) + } +} +impl ServingDrainProof { + pub(super) fn data(&self) -> Result { + let mut decoder = BoundedDecoder::new(&self.0.body, 960)?; + let data = DrainData::decode(&mut decoder)?; + decoder.finish()?; + Ok(data) + } +} +impl WireValue for ServingDrainProof { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.data()?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self(super::super::certificate::CertificateEnvelope::decode(d)?); + value.data()?; + Ok(value) + } +} + +#[cfg(test)] +mod tests { + use super::*; + type TestResult = Result<(), Box>; + fn token() -> ServingToken { + ServingToken { + repository: *uuid::Uuid::new_v4().as_bytes(), + reader: [1; 16], + owner: OwnerFence { + incarnation: IncarnationId::from_bytes([2; 16]), + epoch: 1, + }, + admission_sequence: 1, + generation: 1, + } + } + fn qualify(value: T) -> TestResult { + let mut e = BoundedEncoder::new(1024)?; + value.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 1024)?; + assert_eq!(T::decode(&mut d)?, value); + d.finish()?; + for cut in 0..bytes.len() { + let mut d = BoundedDecoder::new(&bytes[..cut], 1024)?; + assert!(T::decode(&mut d).is_err()); + } + let mut trailing = bytes; + trailing.push(0); + let mut d = BoundedDecoder::new(&trailing, 1024)?; + T::decode(&mut d)?; + assert!(d.finish().is_err()); + Ok(()) + } + #[test] + fn serving_codecs_reject_truncation_trailing_bytes_and_invalid_bounds() -> TestResult { + let token = token(); + qualify(token)?; + qualify(ServingCheck { token, actor: None })?; + qualify(AcquireServingRequest { + repository: token.repository, + reader: token.reader, + actor: Some("reader".into()), + lease_ms: MAX_LEASE_MS, + })?; + qualify(RenewServingRequest { + check: ServingCheck { + token, + actor: Some("reader".into()), + }, + lease_ms: 1, + })?; + for denial in [ + ServingDenial::Unauthorized, + ServingDenial::Conflict, + ServingDenial::Uninitialized, + ServingDenial::Stale, + ServingDenial::Expired, + ServingDenial::Capacity, + ] { + qualify(ServingReply::Denied(denial))?; + qualify(ServingReleaseReply::Denied(denial))?; + } + qualify(ServingReleaseReply::Released)?; + for lease_ms in [0, MAX_LEASE_MS + 1, u64::MAX] { + assert!( + AcquireServingRequest { + repository: token.repository, + reader: token.reader, + actor: None, + lease_ms + } + .encode(&mut BoundedEncoder::new(1024)?) + .is_err() + ); + } + for field in 0..5 { + let mut invalid = token; + match field { + 0 => invalid.reader = [0; 16], + 1 => invalid.owner.epoch = 0, + 2 => invalid.admission_sequence = 0, + 3 => invalid.generation = 0, + _ => invalid.generation = u64::MAX, + } + assert!(invalid.encode(&mut BoundedEncoder::new(1024)?).is_err()); + } + Ok(()) + } + #[test] + fn drain_proof_mac_and_domain_bind_scope_and_exact_pin() -> TestResult { + let data = DrainData { + tenant: [3; 16], + application: [4; 16], + token: token(), + administrator: "owner".into(), + }; + let seed = [5; 32]; + let proof = ServingDrainProof(super::super::super::certificate::CertificateEnvelope::seal( + &data, &seed, + )?); + qualify(proof.clone())?; + assert_eq!(proof.data()?, data); + assert!(proof.0.authenticated(&seed)); + assert!(!proof.0.authenticated(&[6; 32])); + let mut tampered = proof.clone(); + let last = tampered.0.body.len() - 1; + tampered.0.body[last] ^= 1; + assert!(tampered.data().is_ok()); + assert!(!tampered.0.authenticated(&seed)); + let mut other_domain = proof; + other_domain.0.body[4] ^= 1; + assert!(other_domain.data().is_err()); + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/command_owner.rs b/crates/canopy-server/src/packs/publication/serving/command_owner.rs new file mode 100644 index 00000000..ffdadb15 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/command_owner.rs @@ -0,0 +1,168 @@ +//! Registered originals reuse custody history and the common publication queue. +use super::super::custody::{OwnedCustody, RESERVATION, project}; +use super::*; +use cellule_runtime::{Committed, InvocationError, PendingMutation}; + +#[must_use] +pub struct ReadyServingCommand { + client: CellClient, + target: CellTarget, + request: BeginRequest, + original: Arc, + // Only the original local acquisition can hand off physical ownership. + // Restored journal knowledge and renewals cannot recreate it. + local_acquisition: bool, + // Renewal is owned physical work from preparation until known disposition. + _guard: Option>, +} +impl ReadyServingCommand { + pub async fn acquire( + client: CellClient, + target: CellTarget, + request: BeginRequest, + identity: MutationIdentity, + authority: PreparationAuthority, + ) -> Result { + request.encode(&mut BoundedEncoder::new(4096)?)?; + if !authority.matches(&target) + || crate::repository_target(target.tenant(), target.application(), request.repository)? + != target + { + return Err(CustodyError::Context); + } + let original = OwnedCustody::prepare( + &client, + &target, + CustodyAction::AcquireServing(request.clone()), + identity, + ) + .await?; + Ok(Self { + client, + target, + request, + original: Arc::new(original), + local_acquisition: true, + _guard: None, + }) + } + pub(super) async fn renew( + client: CellClient, + target: CellTarget, + request: RenewServingRequest, + request_digest: [u8; 32], + identity: MutationIdentity, + guard: Arc, + ) -> Result { + let action = CustodyAction::RenewServing { + request, + request_digest, + }; + let context = context(&action)?; + let original = OwnedCustody::prepare(&client, &target, action, identity).await?; + Ok(Self { + client, + target, + request: context, + original: Arc::new(original), + local_acquisition: false, + _guard: Some(guard), + }) + } + /// Reconstruct the recorded serving original. A historical grant remains + /// knowledge; ServingPin construction separately rechecks physical authority. + pub async fn restore( + client: CellClient, + target: CellTarget, + reader: [u8; 16], + authority: PreparationAuthority, + ) -> Result { + if !authority.matches(&target) { + return Err(CustodyError::Context); + } + let original = + OwnedCustody::restore_for(&client, &target, CustodyPurpose::Serving, reader).await?; + let request = context(&original.action()?)?; + Ok(Self { + client, + target, + request, + original: Arc::new(original), + local_acquisition: false, + _guard: None, + }) + } + pub fn evidence(&self) -> &PendingMutation { + self.original.evidence() + } + /// Retain this accepted local acquisition even if its lease expired or the + /// requesting account lost Read. Every I/O still checks fresh Read/expiry; + /// this handoff permits safe physical ownership and authenticated cleanup. + pub async fn retain_acquisition( + &self, + context: ServingContext, + ) -> Result { + if !self.local_acquisition || self.target != context.target_for_handoff() { + return Err(ServingReadError::Context); + } + ServingPin::retain_original(context, self.original.clone()).await + } + pub(in crate::packs::publication) fn reservation(&self) -> u64 { + RESERVATION + } + pub(in crate::packs::publication) fn dispatch_copy(&self) -> Self { + Self { + client: self.client.clone(), + target: self.target.clone(), + request: self.request.clone(), + original: self.original.clone(), + local_acquisition: self.local_acquisition, + _guard: self._guard.clone(), + } + } + pub(in crate::packs::publication) fn context( + &self, + ) -> (&CellClient, &CellTarget, BeginRequest) { + (&self.client, &self.target, self.request.clone()) + } + pub(in crate::packs::publication) fn pending(&self) -> PublicationError { + PublicationError::ServingCommand(InvocationError::Pending(Box::new( + self.evidence().clone(), + ))) + } + pub(in crate::packs::publication) async fn dispatch( + self, + recover: bool, + fault: u8, + ) -> Result, PublicationError> { + let result = self + .original + .invoke(&self.client, recover, fault, || Ok(())) + .await + .map_err(|source| PublicationError::Custody { + evidence: Box::new(self.evidence().clone()), + source: Box::new(source), + })?; + project(result, |reply| match reply { + CustodyReply::Serving(reply) => Some(reply), + _ => None, + }) + .map_err(PublicationError::ServingCommand) + } +} +fn context(action: &CustodyAction) -> Result { + match action { + CustodyAction::AcquireServing(request) => Ok(request.clone()), + CustodyAction::RenewServing { + request, + request_digest, + } => Ok(BeginRequest { + repository: request.check.token.repository, + operation: request.check.token.reader, + request_digest: *request_digest, + actor: request.check.actor.clone().ok_or(CustodyError::Context)?, + lease_ms: request.lease_ms, + }), + _ => Err(CustodyError::Context), + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/commands.rs b/crates/canopy-server/src/packs/publication/serving/commands.rs new file mode 100644 index 00000000..a3f245d7 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/commands.rs @@ -0,0 +1,315 @@ +use super::super::sql::*; +use super::*; +use crate::ReadIdentity; +const ROW: &str = "SELECT incarnation,admission_sequence,owner_epoch,generation,expires_at_ms FROM catalog_serving_pins WHERE reader=?1"; +fn access(actor: &Option) -> cellule_runtime::Result { + let actor = actor + .as_deref() + .map_or(ReadIdentity::Anonymous, ReadIdentity::Account); + actor.validate()?; + Ok(statement( + &format!("SELECT 1 WHERE {}", crate::access::READ_ACCESS), + vec![actor.parameter()], + )) +} +fn context(target: &CellTarget, repository: [u8; 16]) -> cellule_runtime::Result { + Ok(*target == crate::repository_target(target.tenant(), target.application(), repository)?) +} +fn row(sets: &[SqlResultSet], token: ServingToken) -> cellule_runtime::Result> { + let Some( + [ + incarnation, + sequence, + epoch, + generation, + SqlValue::Integer(expires), + ], + ) = rows(sets)?.first().map(Vec::as_slice) + else { + if rows(sets)?.is_empty() { + return Ok(None); + } + return Err(Error::Command("invalid serving pin row")); + }; + if fixed::<16>(incarnation)? != *token.owner.incarnation.as_bytes() + || unsigned(sequence)? != token.admission_sequence + || u64::from_be_bytes(fixed(epoch)?) != token.owner.epoch + || unsigned(generation)? != token.generation + { + return Ok(None); + } + Ok(Some(*expires)) +} +fn grant( + token: ServingToken, + fact: GenerationFact, + format: ObjectFormat, + now: i64, + expires: i64, +) -> cellule_runtime::Result { + let value = ServingLease { + token, + fact, + format, + observed_at_ms: now, + expires_at_ms: expires, + }; + value.validate()?; + Ok(value) +} +fn denied(reason: ServingDenial) -> CommandResult { + CommandResult::Rejected(ServingReply::Denied(reason)) +} +pub struct AcquireServingPin; +impl Command for AcquireServingPin { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 44; + const CODEC_VERSION: u32 = 1; + type Input = AcquireServingRequest; + type Output = ServingReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + input.encode(&mut BoundedEncoder::new(1024)?)?; + if !self::context(context.target(), input.repository)? + || rows(&context.sql(&access(&input.actor)?)?)?.is_empty() + { + return Ok(denied(ServingDenial::Unauthorized)); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + input.repository, + )? + else { + return Ok(denied(ServingDenial::Unauthorized)); + }; + let fact = super::super::commands::fact(context, input.repository, format, None)?; + if fact.generation == 0 || fact.catalog.is_none() || fact.refs.is_none() { + return Ok(denied(ServingDenial::Uninitialized)); + } + if !rows(&context.sql(&statement(ROW, vec![blob(input.reader)]))?)?.is_empty() { + return Ok(denied(ServingDenial::Conflict)); + } + let counts = context.sql(&statement( + "SELECT count(*) FROM (SELECT reader FROM catalog_serving_pins LIMIT ?1)", + vec![number(MAX_SERVING_PINS + 1)?], + ))?; + let Some([count]) = rows(&counts)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing serving pin count")); + }; + if unsigned(count)? >= MAX_SERVING_PINS { + return Ok(denied(ServingDenial::Capacity)); + } + let now = now(context.now_ms())?; + let expires = expiry(now, input.lease_ms)?; + let token = ServingToken { + repository: input.repository, + reader: input.reader, + owner: context.owner_fence(), + admission_sequence: context.sequence(), + generation: fact.generation, + }; + token.validate()?; + let changed=context.sql(&statement("INSERT INTO catalog_serving_pins(reader,incarnation,admission_sequence,owner_epoch,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,?6)",vec![blob(input.reader),blob(token.owner.incarnation.as_bytes()),number(token.admission_sequence)?,blob(token.owner.epoch.to_be_bytes()),number(token.generation)?,SqlValue::Integer(expires)]))?; + if changed.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command("serving pin was not inserted")); + } + Ok(CommandResult::Success(ServingReply::Granted(Box::new( + grant(token, fact, format, now, expires)?, + )))) + } +} +pub struct RenewServingPin; +impl Command for RenewServingPin { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 45; + const CODEC_VERSION: u32 = 1; + type Input = RenewServingRequest; + type Output = ServingReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + input.encode(&mut BoundedEncoder::new(1024)?)?; + let token = input.check.token; + if !self::context(context.target(), token.repository)? + || rows(&context.sql(&access(&input.check.actor)?)?)?.is_empty() + { + return Ok(denied(ServingDenial::Unauthorized)); + } + if token.owner != context.owner_fence() { + return Ok(denied(ServingDenial::Stale)); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + token.repository, + )? + else { + return Ok(denied(ServingDenial::Unauthorized)); + }; + let Some(expires) = row( + &context.sql(&statement(ROW, vec![blob(token.reader)]))?, + token, + )? + else { + return Ok(denied(ServingDenial::Conflict)); + }; + let now = now(context.now_ms())?; + if expires <= now { + return Ok(denied(ServingDenial::Expired)); + } + let expires = expiry(now, input.lease_ms)?.max(expires); + let changed = context.sql(&statement( + "UPDATE catalog_serving_pins SET expires_at_ms=?1 WHERE reader=?2", + vec![SqlValue::Integer(expires), blob(token.reader)], + ))?; + if changed.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command("serving pin was not renewed")); + } + let fact = super::super::commands::fact( + context, + token.repository, + format, + Some(token.generation), + )?; + Ok(CommandResult::Success(ServingReply::Granted(Box::new( + grant(token, fact, format, now, expires)?, + )))) + } +} +pub struct CheckServingPin; +impl Query for CheckServingPin { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 47; + const CODEC_VERSION: u32 = 1; + type Input = ServingCheck; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + input: Self::Input, + ) -> cellule_runtime::Result { + input.encode(&mut BoundedEncoder::new(1024)?)?; + let token = input.token; + // QueryContext is scoped by its trusted CellClient capability. The + // service additionally verifies actual target/owner before artifact I/O. + if rows(&context.sql(&access(&input.actor)?)?)?.is_empty() { + return Ok(None); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + token.repository, + )? + else { + return Ok(None); + }; + let Some(expires) = row( + &context.sql(&statement(ROW, vec![blob(token.reader)]))?, + token, + )? + else { + return Ok(None); + }; + let now = now(context.now_ms())?; + if expires <= now { + return Ok(None); + } + let fact = generation( + &context.sql(&statement(GENERATION, vec![number(token.generation)?]))?, + token.repository, + format, + )?; + Ok(Some(grant(token, fact, format, now, expires)?)) + } +} +pub struct SelectServingGeneration; +impl Query for SelectServingGeneration { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 48; + const CODEC_VERSION: u32 = 1; + type Input = ServingSelection; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + input: Self::Input, + ) -> cellule_runtime::Result { + input.encode(&mut BoundedEncoder::new(1024)?)?; + if rows(&context.sql(&access(&input.actor)?)?)?.is_empty() { + return Ok(None); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + input.repository, + )? + else { + return Ok(None); + }; + // Both immutable roots come from one indexed head observation. A caller + // must acquire/check its own exact serving retention before artifact I/O. + let fact = generation( + &context.sql(&statement(CURRENT, vec![]))?, + input.repository, + format, + )?; + if fact.generation == 0 || fact.catalog.is_none() || fact.refs.is_none() { + return Ok(None); + } + Ok(Some(fact)) + } +} +pub struct ReleaseServingPin; +impl Command for ReleaseServingPin { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 46; + const CODEC_VERSION: u32 = 1; + type Input = ServingDrainProof; + type Output = ServingReleaseReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + let data = input.data()?; + let reject = |reason| Ok(CommandResult::Rejected(ServingReleaseReply::Denied(reason))); + if data.tenant != *context.target().tenant().as_bytes() + || data.application != *context.target().application().as_bytes() + || data.token.owner != context.owner_fence() + || super::super::commands::authorized( + context, + data.token.repository, + &data.administrator, + TokenScope::Admin, + )? + .is_none() + { + return reject(ServingDenial::Unauthorized); + } + let seeds = context.sql(&statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ))?; + let Some([seed]) = rows(&seeds)?.first().map(Vec::as_slice) else { + return Err(Error::Command("serving pin seed absent")); + }; + if !input.0.authenticated(&fixed(seed)?) { + return reject(ServingDenial::Unauthorized); + } + if row( + &context.sql(&statement(ROW, vec![blob(data.token.reader)]))?, + data.token, + )? + .is_none() + { + return reject(ServingDenial::Conflict); + } + // Expiry does not remove this root. Only an authenticated drained owner + // can release it; old workers may still own artifacts after their lease. + let changed = context.sql(&statement( + "DELETE FROM catalog_serving_pins WHERE reader=?1", + vec![blob(data.token.reader)], + ))?; + if changed.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command("serving pin was not released")); + } + Ok(CommandResult::Success(ServingReleaseReply::Released)) + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/lifecycle.rs b/crates/canopy-server/src/packs/publication/serving/lifecycle.rs new file mode 100644 index 00000000..4cdbe315 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/lifecycle.rs @@ -0,0 +1,725 @@ +//! A generation's producer survives callers and owns renewal through real drain. +use super::*; +use crate::admission::AdmissionPermit; +use cellule_runtime::InvocationError; +use std::sync::{Mutex, Weak}; +use tokio::{ + sync::watch, + time::{Duration, Instant}, +}; +use tokio_util::task::TaskTracker; + +#[derive(Debug, thiserror::Error)] +pub enum ServingOwnerError { + #[error("serving owner read failed")] + Read(#[from] ServingReadError), + #[error("serving owner clock failed")] + Clock(#[source] Box), + #[error("serving owner scheduling failed")] + Schedule(#[from] PublicationScheduleError), + #[error("serving owner command failed")] + Command(#[source] Arc), + #[error("serving owner custody failed")] + Custody(#[source] Box), + #[error("serving owner invariant failed")] + Context, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ServingOwnerPhase { + Acquiring, + Ready, + Draining, + Released, + Denied, +} +#[derive(Clone, Debug)] +pub struct ServingOwnerStats { + pub phase: ServingOwnerPhase, + pub token: Option, + pub renewals: u64, + pub retries: u64, + pub last_error: Option>, +} +struct Control { + closed: bool, + paused: bool, + borrowers: usize, + pin: Option, +} +struct Inner { + context: ServingContext, + coordinator: PublicationCoordinator, + request: BeginRequest, + control: Mutex, + driver: tokio::sync::Mutex, + changed: tokio::sync::Notify, + tasks: TaskTracker, + updates: watch::Sender, + permit: Mutex>, + #[cfg(test)] + fault: std::sync::atomic::AtomicU8, + #[cfg(test)] + recovery_gate: tokio::sync::Mutex>, +} +#[cfg(test)] +struct RecoveryGate { + entered: tokio::sync::oneshot::Sender<()>, + proceed: tokio::sync::oneshot::Receiver<()>, +} +struct Lifetime(Weak); +impl Drop for Lifetime { + fn drop(&mut self) { + if let Some(inner) = self.0.upgrade() { + inner.close(); + } + } +} +/// Clone shares one producer, one pin and one physical drain. Last handle loss +/// stops new borrows; detached workers keep their private ownership until drain. +#[derive(Clone)] +#[must_use] +pub struct ServingOwner { + inner: Arc, + _lifetime: Arc, +} +/// Joining an existing producer does not keep its admission lifetime open. +/// Residency can observe last-handle cleanup without becoming a producer. +#[derive(Clone)] +pub struct ServingDrainObserver { + tasks: TaskTracker, + stats: watch::Receiver, +} +impl ServingDrainObserver { + pub async fn wait(&self) -> ServingOwnerStats { + self.tasks.close(); + self.tasks.wait().await; + self.stats.borrow().clone() + } +} +struct Borrow { + inner: Arc, + _permit: AdmissionPermit, +} +impl Drop for Borrow { + fn drop(&mut self) { + let mut state = self.inner.control.lock().expect("serving owner control"); + state.borrowers -= 1; + drop(state); + self.inner.changed.notify_waiters(); + } +} +/// Its private guard survives snapshot clones. No raw pin escapes the producer. +#[derive(Clone)] +pub struct ServingSnapshot { + pin: ServingPin, + actor: Option, + _borrow: Arc, +} +impl ServingSnapshot { + pub(crate) async fn resolve_refs( + &self, + names: &[String], + ) -> Result, ServingReadError> { + self.pin.resolve_refs(self.actor.clone(), names).await + } + /// Internal write-preparation inputs. Publication still needs a separately + /// authorized staged producer; this API grants no forward membership proof. + pub(crate) async fn native_base( + &self, + ) -> Result { + self.pin.native_base(self.actor.clone(), self.clone()).await + } + /// Build a complete forward closure from certified roots. Callers choosing + /// fetch roots must independently bind them to this snapshot's live refs. + pub async fn workspace( + &self, + roots: &[crate::ObjectId], + limits: WorkspaceLimits, + ) -> Result { + self.pin + .workspace(self.actor.clone(), Some(roots), limits, self.clone()) + .await + } + /// Build native refs and their exact forward closure from this accepted joint + /// generation. No live SQL refs or caller-supplied object roots are consulted. + pub async fn ref_workspace( + &self, + limits: WorkspaceLimits, + ) -> Result { + self.pin + .workspace(self.actor.clone(), None, limits, self.clone()) + .await + } + pub fn fact(&self) -> GenerationFact { + self.pin.fact() + } + pub async fn headers( + &self, + ids: &[crate::ObjectId], + ) -> Result>, ServingReadError> { + self.pin.headers(self.actor.clone(), ids).await + } + pub async fn edges_page( + &self, + ids: &[crate::ObjectId], + after: Option<(crate::ObjectId, crate::ObjectId)>, + ) -> Result { + self.pin.edges_page(self.actor.clone(), ids, after).await + } + pub async fn body( + &self, + oid: crate::ObjectId, + limit: usize, + ) -> Result>, ServingReadError> { + self.pin.body(self.actor.clone(), oid, limit).await + } + pub async fn resolve_ref( + &self, + reference: Option<&str>, + ) -> Result { + self.pin.resolve_ref(self.actor.clone(), reference).await + } + pub async fn refs_page( + &self, + after: &str, + generation: Option, + live_only: bool, + ) -> Result { + self.pin + .refs_page(self.actor.clone(), after, generation, live_only) + .await + } +} +enum Original { + Command(Arc), + Release(Arc), +} +impl Original { + fn copy(&self) -> ReadyPublication { + match self { + Self::Command(value) => value.dispatch_copy().into(), + Self::Release(value) => value.dispatch_copy().into(), + } + } +} +#[derive(Clone, Copy, PartialEq, Eq)] +enum Kind { + Acquire, + Renew, + Release, +} +struct Pending { + kind: Kind, + original: Original, + ticket: Option, +} +struct Driver { + plan: Option<(Kind, MutationIdentity)>, + pending: Option, + next_renewal: Instant, + done: bool, + renewal_denied: bool, +} +impl ServingOwner { + pub async fn start( + context: ServingContext, + coordinator: PublicationCoordinator, + request: BeginRequest, + identity: MutationIdentity, + ) -> Result { + let mut bytes = BoundedEncoder::new(4096).map_err(ServingReadError::Codec)?; + request + .encode(&mut bytes) + .map_err(ServingReadError::Codec)?; + if coordinator.target() != &context.target_for_handoff() + || crate::repository_target( + coordinator.target().tenant(), + coordinator.target().application(), + request.repository, + ) + .map_err(ServingReadError::Capability)? + != *coordinator.target() + { + return Err(ServingOwnerError::Context); + } + let permit = context.admit_owner(&request.actor).await?; + let (updates, _) = watch::channel(ServingOwnerStats { + phase: ServingOwnerPhase::Acquiring, + token: None, + renewals: 0, + retries: 0, + last_error: None, + }); + let inner = Arc::new(Inner { + context, + coordinator, + request, + control: Mutex::new(Control { + closed: false, + paused: false, + borrowers: 0, + pin: None, + }), + driver: tokio::sync::Mutex::new(Driver { + plan: Some((Kind::Acquire, identity)), + pending: None, + next_renewal: Instant::now(), + done: false, + renewal_denied: false, + }), + changed: tokio::sync::Notify::new(), + tasks: TaskTracker::new(), + updates, + permit: Mutex::new(Some(permit)), + #[cfg(test)] + fault: std::sync::atomic::AtomicU8::new(0), + #[cfg(test)] + recovery_gate: tokio::sync::Mutex::new(None), + }); + inner.tasks.spawn(supervise(inner.clone())); + Ok(Self { + _lifetime: Arc::new(Lifetime(Arc::downgrade(&inner))), + inner, + }) + } + pub fn stats(&self) -> ServingOwnerStats { + self.inner.updates.borrow().clone() + } + pub(super) fn is_drained(&self) -> bool { + self.inner.tasks.is_empty() + } + pub(super) fn retire_if_idle(&self) -> bool { + let mut state = self.inner.control.lock().expect("serving owner control"); + if state.borrowers != 0 { + return false; + } + state.closed = true; + drop(state); + self.inner.changed.notify_waiters(); + true + } + /// A nonwaiting handshake: the worker cannot create another original after + /// its driver lock has been observed while paused. Busy originals stay owned. + pub(super) fn pause_for_drain(&self) -> Option> { + self.inner + .control + .lock() + .expect("serving owner control") + .paused = true; + let Ok(driver) = self.inner.driver.try_lock() else { + return None; + }; + if driver.done { + return Some(None); + } + let state = self.inner.control.lock().expect("serving owner control"); + if state.closed || state.borrowers != 0 || driver.pending.is_some() { + return None; + } + let pin = state.pin.as_ref()?; + pin.workers_idle().then_some(Some(pin.token())) + } + pub(super) fn resume(&self) { + self.inner + .control + .lock() + .expect("serving owner control") + .paused = false; + self.inner.changed.notify_waiters(); + } + pub fn drain_observer(&self) -> ServingDrainObserver { + ServingDrainObserver { + tasks: self.inner.tasks.clone(), + stats: self.inner.updates.subscribe(), + } + } + #[cfg(test)] + pub(in crate::packs::publication) fn fault_for_test(&self, point: u8) { + self.inner + .fault + .store(point, std::sync::atomic::Ordering::Release); + } + #[cfg(test)] + pub(in crate::packs::publication) async fn pause_recovery_for_test( + &self, + ) -> ( + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + ) { + let (proceed, wait) = tokio::sync::oneshot::channel(); + let (entered, observed) = tokio::sync::oneshot::channel(); + *self.inner.recovery_gate.lock().await = Some(RecoveryGate { + entered, + proceed: wait, + }); + (proceed, observed) + } + pub fn close(&self) { + self.inner.close(); + } + pub async fn close_and_drain(&self) -> ServingOwnerStats { + self.close(); + self.inner.tasks.close(); + self.inner.tasks.wait().await; + self.stats() + } + /// Waiters and returned borrows share a separate bounded node/account + /// budget; long-lived snapshots cannot consume every physical I/O slot. + pub async fn snapshot( + &self, + actor: Option, + ) -> Result { + let permit = self.inner.context.admit_snapshot(&actor).await?; + self.snapshot_admitted(actor, permit).await + } + pub(super) async fn snapshot_admitted( + &self, + actor: Option, + permit: AdmissionPermit, + ) -> Result { + let inner = self.inner.clone(); + self.inner + .context + .tasks() + .spawn(async move { + let mut permit = Some(permit); + loop { + let changed = inner.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + let acquired = { + let mut state = inner.control.lock().expect("serving owner control"); + if state.closed || state.paused { + return Err(ServingReadError::Inactive); + } + if let Some(pin) = state.pin.clone() { + state.borrowers += 1; + Some(ServingSnapshot { + pin, + actor: actor.clone(), + _borrow: Arc::new(Borrow { + inner: inner.clone(), + _permit: permit.take().expect("snapshot admission"), + }), + }) + } else { + None + } + }; + if let Some(snapshot) = acquired { + snapshot.pin.authorize(actor).await?; + return Ok(snapshot); + } + changed.await; + } + }) + .await? + } +} +impl Inner { + #[cfg(test)] + fn failpoint(&self, point: u8) { + if self + .fault + .compare_exchange( + point, + 0, + std::sync::atomic::Ordering::AcqRel, + std::sync::atomic::Ordering::Acquire, + ) + .is_ok() + { + panic!("injected serving producer panic at {point}"); + } + } + fn close(&self) { + self.control.lock().expect("serving owner control").closed = true; + self.changed.notify_waiters(); + } + fn pin(&self) -> Option { + self.control + .lock() + .expect("serving owner control") + .pin + .clone() + } + fn update(&self, change: impl FnOnce(&mut ServingOwnerStats)) { + self.updates.send_modify(change); + } + async fn step(&self) -> Result { + // Plans, exact bodies and held tickets live outside the supervised task. + // Panic during any awaited provider work cannot invent a replacement. + let mut driver = self.driver.lock().await; + if driver.done { + return Ok(true); + } + if driver.pending.is_none() { + let (closed, borrowers, pin) = { + let state = self.control.lock().expect("serving owner control"); + if state.paused && !state.closed { + return Ok(false); + } + (state.closed, state.borrowers, state.pin.clone()) + }; + // A factory has not registered or submitted anything. Once all + // borrows end, an unbuilt renewal plan can yield to physical drain. + // An already built/admitted original never takes this shortcut. + if closed + && borrowers == 0 + && driver + .plan + .as_ref() + .is_some_and(|(kind, _)| *kind == Kind::Renew) + { + driver.plan.take(); + } + if driver.plan.is_none() { + let kind = if closed && borrowers == 0 { + Kind::Release + } else if driver.renewal_denied { + return Ok(false); + } else if Instant::now() >= driver.next_renewal { + Kind::Renew + } else { + return Ok(false); + }; + if pin.is_none() { + return Err(ServingOwnerError::Context); + } + driver.plan = Some(( + kind, + crate::server::mutation_identity() + .map_err(|error| ServingOwnerError::Clock(Box::new(error)))?, + )); + } + let (kind, identity) = driver.plan.expect("owned factory plan"); + let original = match kind { + Kind::Acquire => Original::Command(Arc::new( + ReadyServingCommand::acquire( + self.context.client_for_owner(), + self.context.target_for_handoff(), + self.request.clone(), + identity, + self.context.authority_for_owner(), + ) + .await + .map_err(|error| ServingOwnerError::Custody(Box::new(error)))?, + )), + Kind::Renew => Original::Command(Arc::new( + pin.ok_or(ServingOwnerError::Context)? + .ready_renew( + self.request.actor.clone(), + self.request.request_digest, + identity, + self.request.lease_ms, + ) + .await?, + )), + Kind::Release => { + self.update(|stats| stats.phase = ServingOwnerPhase::Draining); + Original::Release(Arc::new( + pin.ok_or(ServingOwnerError::Context)? + .ready_release(identity) + .await?, + )) + } + }; + driver.pending = Some(Pending { + kind, + original, + ticket: None, + }); + #[cfg(test)] + self.failpoint(1); + } + let pending = driver.pending.as_mut().expect("owned original"); + if pending.ticket.is_none() { + match self.coordinator.try_reserve(pending.original.copy()) { + Ok(ticket) => { + pending.ticket = Some(ticket); + #[cfg(test)] + self.failpoint(2); + } + Err(refused) => return Err(refused.reason.into()), + } + } + let ticket = pending.ticket.as_ref().expect("owned held ticket"); + match ticket.state() { + PublicationState::Held => { + ticket.activate().await?; + Ok(false) + } + PublicationState::Queued | PublicationState::Running => { + ticket.wait().await; + Ok(false) + } + PublicationState::Uncertain(_) => { + #[cfg(test)] + if let Some(gate) = self.recovery_gate.lock().await.take() { + let _ = gate.entered.send(()); + let _ = gate.proceed.await; + } + ticket.recover().await?; + Ok(false) + } + PublicationState::Discarded => Err(ServingOwnerError::Context), + PublicationState::Finished(Err(error)) => { + let denied = matches!( + &*error, + PublicationError::ServingCommand(InvocationError::Rejected(_)) + ) || matches!(&*error, PublicationError::Custody { source, .. } if + matches!(&**source, CustodyError::Stopped(_)) || + matches!(&**source, CustodyError::Registration(error) if matches!(&**error, InvocationError::Rejected(_)))); + if denied && pending.kind != Kind::Release { + let has_pin = self.pin().is_some(); + self.close(); + driver.pending.take(); + driver.plan.take(); + driver.renewal_denied = true; + if !has_pin { + self.update(|stats| { + stats.phase = ServingOwnerPhase::Denied; + stats.last_error = Some(Arc::new(ServingOwnerError::Command(error))); + }); + driver.done = true; + return Ok(true); + } + return Ok(false); + } + if pending.kind == Kind::Release + && matches!( + &*error, + PublicationError::ServingRelease(InvocationError::Rejected(_)) + ) + { + // An immutable known denial releases nothing. A new proof + // can be prepared only after this original is settled. + driver.pending.take(); + driver.plan.take(); + return Err(ServingOwnerError::Command(error)); + } + // A proven non-execution retries the original; its body never + // changes because the factory plan remains retained. + pending.ticket.take(); + Err(ServingOwnerError::Command(error)) + } + PublicationState::Finished(Ok(PublicationOutcome::ServingRelease(value))) + if pending.kind == Kind::Release + && value.output == ServingReleaseReply::Released => + { + #[cfg(test)] + self.failpoint(5); + driver.pending.take(); + driver.plan.take(); + self.control + .lock() + .expect("serving owner control") + .pin + .take(); + driver.done = true; + self.update(|stats| stats.phase = ServingOwnerPhase::Released); + Ok(true) + } + PublicationState::Finished(Ok(PublicationOutcome::ServingCommand(value))) + if pending.kind != Kind::Release => + { + let ServingReply::Granted(lease) = value.output else { + return Err(ServingOwnerError::Context); + }; + let pin = if pending.kind == Kind::Acquire { + let Original::Command(original) = &pending.original else { + return Err(ServingOwnerError::Context); + }; + if let Some(pin) = self.pin() { + pin + } else { + let pin = original.retain_acquisition(self.context.clone()).await?; + self.control.lock().expect("serving owner control").pin = Some(pin.clone()); + pin + } + } else { + self.pin().ok_or(ServingOwnerError::Context)? + }; + if lease.token != pin.token() || lease.fact != pin.fact() { + return Err(ServingOwnerError::Context); + } + #[cfg(test)] + self.failpoint(if pending.kind == Kind::Acquire { 3 } else { 4 }); + let renewed = pending.kind == Kind::Renew; + driver.pending.take(); + driver.plan.take(); + self.update(|stats| { + stats.token = Some(pin.token()); + if renewed { + stats.renewals = stats.renewals.saturating_add(1); + } + }); + match pin.authorize(Some(self.request.actor.clone())).await { + Ok(deadline) => { + driver.next_renewal = + Instant::now() + deadline.saturating_duration_since(Instant::now()) / 3; + self.update(|stats| stats.phase = ServingOwnerPhase::Ready); + self.changed.notify_waiters(); + } + Err(error) => { + self.close(); + return Err(error.into()); + } + } + Ok(false) + } + _ => Err(ServingOwnerError::Context), + } + } +} +async fn run(inner: Arc) { + loop { + let changed = inner.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + match inner.step().await { + Ok(true) => { + inner.permit.lock().expect("serving owner admission").take(); + inner.changed.notify_waiters(); + return; + } + Ok(false) => {} + Err(error) => { + if matches!( + &error, + ServingOwnerError::Read( + ServingReadError::Inactive + | ServingReadError::Authority(PreparationBaseError::Inactive) + ) + ) { + inner.close(); + // No original was submitted if its factory failed. Wait + // for borrowers instead of churning unbuildable renewals. + let mut driver = inner.driver.lock().await; + if driver.pending.is_none() && inner.pin().is_some() { + driver.plan.take(); + driver.renewal_denied = true; + } + } + inner.update(|stats| { + stats.retries = stats.retries.saturating_add(1); + stats.last_error = Some(Arc::new(error)); + }); + } + } + tokio::select! { _ = changed => {}, _ = tokio::time::sleep(Duration::from_millis(100)) => {} } + } +} +async fn supervise(inner: Arc) { + loop { + match tokio::spawn(run(inner.clone())).await { + Ok(()) => return, + Err(error) => inner.update(|stats| { + stats.retries = stats.retries.saturating_add(1); + stats.last_error = Some(Arc::new(ServingOwnerError::Read(ServingReadError::Task( + error, + )))); + }), + } + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/ownership.rs b/crates/canopy-server/src/packs/publication/serving/ownership.rs new file mode 100644 index 00000000..03ac6237 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/ownership.rs @@ -0,0 +1,56 @@ +//! One physical drain owner per exact pin across every context in this process. +//! A weak entry survives while any worker, pin or retained release owns its Arc. +use super::*; +use std::{ + collections::HashMap, + sync::{Mutex, OnceLock, Weak}, +}; + +pub const MAX_SERVING_OWNERS: usize = 4096; +#[derive(Clone, Copy, PartialEq, Eq, Hash)] +struct Key { + tenant: [u8; 16], + application: [u8; 16], + repository: [u8; 16], + reader: [u8; 16], + incarnation: [u8; 16], + epoch: u64, + sequence: u64, +} +static OWNERS: OnceLock>>> = OnceLock::new(); + +pub(super) fn reserve( + target: &CellTarget, + token: ServingToken, +) -> Result, ServingReadError> { + token.validate()?; + if crate::repository_target(target.tenant(), target.application(), token.repository)? != *target + { + return Err(ServingReadError::Context); + } + let key = Key { + tenant: *target.tenant().as_bytes(), + application: *target.application().as_bytes(), + repository: token.repository, + reader: token.reader, + incarnation: *token.owner.incarnation.as_bytes(), + epoch: token.owner.epoch, + sequence: token.admission_sequence, + }; + let mut owners = OWNERS + .get_or_init(|| Mutex::new(HashMap::new())) + .lock() + .expect("serving ownership"); + owners.retain(|_, owner| owner.strong_count() != 0); + if owners.contains_key(&key) { + return Err(ServingReadError::AlreadyOwned); + } + if owners.len() >= MAX_SERVING_OWNERS { + return Err(ServingReadError::Capability(Error::Capacity( + "node serving owners", + ))); + } + let owner = Arc::new(()); + owners.insert(key, Arc::downgrade(&owner)); + Ok(owner) +} diff --git a/crates/canopy-server/src/packs/publication/serving/pool.rs b/crates/canopy-server/src/packs/publication/serving/pool.rs new file mode 100644 index 00000000..770d74e0 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/pool.rs @@ -0,0 +1,297 @@ +//! Bounded resident generation ownership; observers cannot abandon acquisitions. +use super::*; +use crate::admission::AdmissionPermit; +use std::sync::{ + Weak, + atomic::{AtomicBool, Ordering}, +}; +use tokio::sync::Mutex; +use tokio_util::{sync::CancellationToken, task::TaskTracker}; + +pub const MAX_SERVING_GENERATIONS: u8 = 4; +#[derive(Clone, Copy, Debug)] +pub struct ServingPoolLimits { + pub generations: u8, + pub lease_ms: u64, +} +impl Default for ServingPoolLimits { + fn default() -> Self { + Self { + generations: MAX_SERVING_GENERATIONS, + lease_ms: DEFAULT_LEASE_MS, + } + } +} +struct Slot { + requested_generation: u64, + touched: u64, + owner: ServingOwner, +} +struct State { + closed: bool, + clock: u64, + slots: Vec, +} +struct Inner { + context: ServingContext, + coordinator: PublicationCoordinator, + limits: ServingPoolLimits, + state: Mutex, + paused: AtomicBool, + stop: CancellationToken, + requests: TaskTracker, + drain: TaskTracker, +} +struct Lifetime(Weak); +impl Drop for Lifetime { + fn drop(&mut self) { + if let Some(inner) = self.0.upgrade() { + inner.stop.cancel(); + } + } +} +/// Clones share one residency lifetime and at most four retained generation owners. +#[derive(Clone)] +#[must_use] +pub struct ServingPool { + inner: Arc, + _lifetime: Arc, +} +struct Resume { + inner: Arc, + owners: Vec, +} +impl Drop for Resume { + fn drop(&mut self) { + for owner in &self.owners { + owner.resume(); + } + self.inner.paused.store(false, Ordering::Release); + } +} +impl ServingPool { + pub fn new( + context: ServingContext, + coordinator: PublicationCoordinator, + limits: ServingPoolLimits, + ) -> Result { + if !(1..=MAX_SERVING_GENERATIONS).contains(&limits.generations) + || !(1000..=MAX_LEASE_MS).contains(&limits.lease_ms) + || coordinator.target() != &context.target_for_handoff() + { + return Err(ServingOwnerError::Context); + } + let inner = Arc::new(Inner { + context, + coordinator, + limits, + state: Mutex::new(State { + closed: false, + clock: 0, + slots: Vec::new(), + }), + paused: AtomicBool::new(false), + stop: CancellationToken::new(), + requests: TaskTracker::new(), + drain: TaskTracker::new(), + }); + let work = inner.clone(); + inner.drain.spawn(async move { + work.stop.cancelled().await; + let owners = { + let mut state = work.state.lock().await; + state.closed = true; + state + .slots + .iter() + .map(|slot| slot.owner.clone()) + .collect::>() + }; + for owner in &owners { + owner.close(); + } + work.requests.close(); + work.requests.wait().await; + // Independent producers keep renewing/draining concurrently. One + // blocked old generation cannot stop another's exact release. + futures_util::future::join_all(owners.iter().map(|owner| owner.close_and_drain())) + .await; + }); + Ok(Self { + _lifetime: Arc::new(Lifetime(Arc::downgrade(&inner))), + inner, + }) + } + pub async fn snapshot( + &self, + actor: Option, + ) -> Result { + if self.inner.stop.is_cancelled() || self.inner.paused.load(Ordering::Acquire) { + return Err(ServingReadError::Inactive.into()); + } + // Admission covers queued selection, acquisition wait and the returned + // borrow. A canceled observer cannot create unbounded detached waiters. + let permit = self.inner.context.admit_snapshot(&actor).await?; + let inner = self.inner.clone(); + self.inner + .requests + .spawn(async move { inner.snapshot(actor, permit).await }) + .await + .map_err(ServingReadError::Task)? + } + pub fn close(&self) { + self.inner.stop.cancel(); + } + pub async fn close_and_drain(&self) { + self.close(); + self.inner.drain.close(); + self.inner.drain.wait().await; + } + /// A private task owns the whole pause/gate/release handshake. Cancellation + /// of this observer neither strands paused owners nor abandons accepted drain. + pub async fn quiesce(&self) -> Result { + let inner = self.inner.clone(); + self.inner + .drain + .spawn(async move { inner.quiesce().await }) + .await + .map_err(ServingReadError::Task)? + } + #[cfg(test)] + pub(in crate::packs::publication) async fn owners_for_test(&self) -> Vec { + self.inner + .state + .lock() + .await + .slots + .iter() + .map(|slot| slot.owner.clone()) + .collect() + } +} +impl Inner { + async fn snapshot( + self: Arc, + actor: Option, + permit: AdmissionPermit, + ) -> Result { + if self.stop.is_cancelled() || self.paused.load(Ordering::Acquire) { + return Err(ServingReadError::Inactive.into()); + } + let selected = self.context.select(actor.clone()).await?; + let owner = { + let mut state = self.state.lock().await; + if state.closed || self.stop.is_cancelled() || self.paused.load(Ordering::Acquire) { + return Err(ServingReadError::Inactive.into()); + } + state.slots.retain(|slot| !slot.owner.is_drained()); + state.clock = state.clock.saturating_add(1); + let touched = state.clock; + if let Some(slot) = state.slots.iter_mut().find(|slot| { + let stats = slot.owner.stats(); + match stats.token { + Some(token) => { + stats.phase == ServingOwnerPhase::Ready + && token.generation == selected.generation + } + None => { + stats.phase == ServingOwnerPhase::Acquiring + && slot.requested_generation == selected.generation + } + } + }) { + slot.touched = touched; + slot.owner.clone() + } else { + if state.slots.len() >= usize::from(self.limits.generations) { + // Keep the closing slot until its real producer finishes. + // Retry is explicit; there is no unbounded retired inventory + // or wait behind old provider I/O inside the pool lock. + let mut order: Vec<_> = (0..state.slots.len()).collect(); + order.sort_by_key(|i| state.slots[*i].touched); + for i in order { + if state.slots[i].owner.retire_if_idle() { + break; + } + } + return Err(ServingReadError::Capability(Error::Capacity( + "repository serving generations", + )) + .into()); + } + let operation = *uuid::Uuid::new_v4().as_bytes(); + let mut digest = blake3::Hasher::new(); + digest.update(b"canopy.serving-pool.v1"); + digest.update(&self.context.repository()); + digest.update(&operation); + let owner = ServingOwner::start( + self.context.clone(), + self.coordinator.clone(), + BeginRequest { + repository: self.context.repository(), + operation, + request_digest: *digest.finalize().as_bytes(), + actor: self.context.administrator().to_owned(), + lease_ms: self.limits.lease_ms, + }, + crate::server::mutation_identity() + .map_err(|error| ServingOwnerError::Clock(Box::new(error)))?, + ) + .await?; + state.slots.push(Slot { + requested_generation: selected.generation, + touched, + owner: owner.clone(), + }); + owner + } + }; + // The accepted acquisition may select a newer fact than the observation. + // Return its fact; never label that capability with the requested hint. + Ok(owner.snapshot_admitted(actor, permit).await?) + } + async fn quiesce(self: Arc) -> Result { + let state = self.state.lock().await; + if state.closed { + let drained = + self.requests.is_empty() && state.slots.iter().all(|slot| slot.owner.is_drained()); + drop(state); + return Ok(drained && self.coordinator.close_if_idle().await); + } + if self.paused.swap(true, Ordering::AcqRel) { + return Ok(false); + } + let pause = Resume { + inner: self.clone(), + owners: state.slots.iter().map(|slot| slot.owner.clone()).collect(), + }; + drop(state); + let mut tokens = Vec::new(); + for owner in &pause.owners { + match owner.pause_for_drain() { + Some(Some(token)) => tokens.push(token), + Some(None) => {} + None => return Ok(false), + } + } + let Some(gate) = self.coordinator.reserve_serving_drain(&tokens).await? else { + return Ok(false); + }; + // No original can be built between the handshake and exclusive gate. + // Queries already admitted finish without creating another slot. + self.stop.cancel(); + for owner in &pause.owners { + owner.close(); + } + self.requests.close(); + self.requests.wait().await; + futures_util::future::join_all(pause.owners.iter().map(|owner| owner.close_and_drain())) + .await; + while !gate.close_if_drained().await { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + // Private supervisor can finish concurrently; do not wait its shared + // tracker here, which also owns this quiesce operation. + Ok(true) + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/schema.sql b/crates/canopy-server/src/packs/publication/serving/schema.sql new file mode 100644 index 00000000..58b5137b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/schema.sql @@ -0,0 +1,23 @@ +-- Bounded read retention, not a creating namespace or historical object table. +-- Expired readers stop serving but remain GC roots until physical work drains. +CREATE TABLE catalog_serving_pins ( + reader BLOB PRIMARY KEY CHECK(typeof(reader)='blob' AND length(reader)=16 AND reader!=zeroblob(16)), + incarnation BLOB NOT NULL CHECK(typeof(incarnation)='blob' AND length(incarnation)=16), + admission_sequence INTEGER NOT NULL CHECK(typeof(admission_sequence)='integer' AND admission_sequence>0), + owner_epoch BLOB NOT NULL CHECK(typeof(owner_epoch)='blob' AND length(owner_epoch)=8 AND owner_epoch!=zeroblob(8)), + generation INTEGER NOT NULL REFERENCES catalog_generations(generation) CHECK(typeof(generation)='integer' AND generation>0), + expires_at_ms INTEGER NOT NULL CHECK(typeof(expires_at_ms)='integer' AND expires_at_ms>=0), + UNIQUE(incarnation,admission_sequence) +) WITHOUT ROWID; +CREATE INDEX catalog_serving_pins_by_generation ON catalog_serving_pins(generation); +CREATE TRIGGER catalog_serving_pin_not_replaced BEFORE INSERT ON catalog_serving_pins +WHEN EXISTS(SELECT 1 FROM catalog_serving_pins WHERE reader=NEW.reader OR (incarnation=NEW.incarnation AND admission_sequence=NEW.admission_sequence)) +BEGIN SELECT RAISE(ABORT,'serving pin cannot be replaced'); END; +CREATE TRIGGER catalog_serving_pin_identity_immutable BEFORE UPDATE ON catalog_serving_pins +WHEN NEW.reader IS NOT OLD.reader OR NEW.incarnation IS NOT OLD.incarnation + OR NEW.admission_sequence IS NOT OLD.admission_sequence OR NEW.owner_epoch IS NOT OLD.owner_epoch + OR NEW.generation IS NOT OLD.generation OR NEW.expires_at_ms=4096 +BEGIN SELECT RAISE(ABORT,'serving pin capacity'); END; diff --git a/crates/canopy-server/src/packs/publication/serving/session.rs b/crates/canopy-server/src/packs/publication/serving/session.rs new file mode 100644 index 00000000..2e1a51d6 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session.rs @@ -0,0 +1,612 @@ +use super::*; +use crate::{ + ReadIdentity, + admission::AccountAdmission, + packs::{ + catalog::{CatalogFiles, CatalogIndexes, CatalogReader}, + metadata::{ObjectHeader, PAGE_OBJECTS}, + }, +}; +use cellule_runtime::{Committed, InvocationError, Receipt, primitives::sql::SqlCell}; +use std::sync::Mutex; +use tokio::{sync::Notify, time::Instant}; +use tokio_util::{sync::CancellationToken, task::TaskTracker}; +mod body; +mod edges; +mod native_base; +mod workspace; +pub use edges::{MAX_EDGE_PARENTS, ServingEdgePage}; +pub use workspace::{NativeWorkspace, WorkspaceLimits, WorkspaceStats}; +mod handoff; +mod reads; +mod refs; +pub use refs::ResolvedServingRef; + +#[derive(Debug, thiserror::Error)] +pub enum ServingReadError { + #[error("serving pin already has a physical drain owner")] + AlreadyOwned, + #[error("serving read is inactive or unavailable")] + Inactive, + #[error("invalid serving context or budget")] + Context, + #[error("serving refs changed while reading pages")] + Changed, + #[error("serving ref snapshot failed")] + RefSnapshot(#[from] crate::packs::ref_state::RefSnapshotError), + #[error("serving ref index failed")] + Refs(#[from] crate::packs::ref_state::RefStateError), + #[error("serving authority failed")] + Authority(#[from] PreparationBaseError), + #[error("serving query failed")] + Query(#[source] Box>>), + #[error("serving generation selection failed")] + Selection(#[source] Box>>), + #[error("serving metadata failed")] + Metadata(#[from] crate::packs::directory::index::IndexError), + #[error("serving capability failed")] + Capability(#[from] Error), + #[error("serving custody intent failed")] + Custody(#[source] Box), + #[error("serving encoding failed")] + Codec(#[from] CodecError), + #[error("serving body exceeds its read limit")] + TooLarge, + #[error("serving native body failed")] + Native(#[from] crate::packs::catalog::NativeReadError), + #[error("serving worker failed")] + Task(#[from] tokio::task::JoinError), + #[error("serving release proof query failed")] + Proof(#[source] Box>>), + #[error("serving release command preparation failed")] + Prepare(#[source] Box>), +} +#[derive(Clone)] +pub struct ServingReadBudget { + inner: Arc, +} +struct Budget { + admission: AccountAdmission, + owners: AccountAdmission, + snapshots: AccountAdmission, + tasks: TaskTracker, + stop: CancellationToken, +} +impl ServingReadBudget { + pub fn new(limit: u16, tasks: TaskTracker) -> Result { + if !(2..=64).contains(&limit) { + return Err(ServingReadError::Context); + } + Ok(Self { + inner: Arc::new(Budget { + admission: AccountAdmission::new( + usize::from(limit), + "node serving reads", + "account serving reads", + ), + owners: AccountAdmission::new( + usize::from(limit), + "node serving owners", + "account serving owners", + ), + snapshots: AccountAdmission::new( + usize::from(limit), + "node serving snapshots", + "account serving snapshots", + ), + tasks, + stop: CancellationToken::new(), + }), + }) + } + pub fn close(&self) { + self.inner.stop.cancel(); + } +} +/// Trusted service configuration. No decoded catalog or lease DTO supplies +/// closure authority: open reobserves the registered pin through this client. +#[derive(Clone)] +pub struct ServingContext { + client: CellClient, + target: CellTarget, + authority: PreparationAuthority, + indexes: Arc, + files: Arc, + budget: ServingReadBudget, + administrator: String, +} +impl ServingContext { + pub(super) fn repository(&self) -> [u8; 16] { + self.indexes.store().repository() + } + pub(super) fn administrator(&self) -> &str { + &self.administrator + } + pub(super) async fn select( + &self, + actor: Option, + ) -> Result { + if self.budget.inner.stop.is_cancelled() { + return Err(ServingReadError::Inactive); + } + let scope = actor + .as_deref() + .map_or(ReadIdentity::Anonymous, ReadIdentity::Account); + let permit = self.budget.inner.admission.acquire(scope).await?; + let context = self.clone(); + self.tasks() + .spawn(async move { + let _permit = permit; + if context.budget.inner.stop.is_cancelled() { + return Err(ServingReadError::Inactive); + } + context + .client + .query::( + &context.target, + None, + ServingSelection { + repository: context.repository(), + actor, + }, + ) + .await + .map_err(|error| ServingReadError::Selection(Box::new(error)))? + .output + .ok_or(ServingReadError::Inactive) + }) + .await? + } + pub(super) fn client_for_owner(&self) -> CellClient { + self.client.clone() + } + pub(super) fn authority_for_owner(&self) -> PreparationAuthority { + self.authority.clone() + } + pub(super) async fn admit_owner( + &self, + actor: &str, + ) -> Result { + if self.budget.inner.stop.is_cancelled() { + return Err(ServingReadError::Inactive); + } + Ok(self + .budget + .inner + .owners + .acquire(ReadIdentity::Account(actor)) + .await?) + } + pub(super) async fn admit_snapshot( + &self, + actor: &Option, + ) -> Result { + if self.budget.inner.stop.is_cancelled() { + return Err(ServingReadError::Inactive); + } + let actor = actor + .as_deref() + .map_or(ReadIdentity::Anonymous, ReadIdentity::Account); + actor.validate()?; + Ok(self.budget.inner.snapshots.acquire(actor).await?) + } + pub(super) fn tasks(&self) -> TaskTracker { + self.budget.inner.tasks.clone() + } + pub(super) fn target_for_handoff(&self) -> CellTarget { + self.target.clone() + } + pub fn new( + client: CellClient, + target: CellTarget, + authority: PreparationAuthority, + indexes: Arc, + files: Arc, + budget: ServingReadBudget, + administrator: String, + ) -> Result { + validate_component(&administrator)?; + if !authority.matches(&target) + || crate::repository_target( + target.tenant(), + target.application(), + indexes.store().repository(), + )? != target + { + return Err(ServingReadError::Context); + } + Ok(Self { + client, + target, + authority, + indexes, + files, + budget, + administrator, + }) + } +} +#[derive(Clone)] +pub struct ServingPin { + inner: Arc, +} +struct Inner { + // Must outlive every active worker and original release command. + _exclusive: Arc<()>, + context: ServingContext, + lease: ServingLease, + state: Mutex, + changed: Notify, + reader: tokio::sync::Mutex>>, + refs: tokio::sync::OnceCell, + release: tokio::sync::Mutex>, +} +#[derive(Default)] +struct Workers { + closed: bool, + active: usize, + released: bool, +} +pub(super) struct Active(Arc); +impl Drop for Active { + fn drop(&mut self) { + let mut state = self.0.state.lock().expect("serving workers"); + state.active -= 1; + drop(state); + self.0.changed.notify_waiters(); + } +} +struct ReleaseCommand { + command: Arc>, + digest: [u8; 32], +} +impl ServingPin { + pub(super) fn workers_idle(&self) -> bool { + let state = self.inner.state.lock().expect("serving workers"); + !state.closed && state.active == 0 + } + pub(super) async fn authorize( + &self, + actor: Option, + ) -> Result { + Ok(self.inner.observe(actor).await?.1) + } + pub async fn open( + context: ServingContext, + token: ServingToken, + actor: Option, + ) -> Result { + if context.budget.inner.stop.is_cancelled() { + return Err(ServingReadError::Inactive); + } + let scope = actor + .as_deref() + .map_or(ReadIdentity::Anonymous, ReadIdentity::Account); + let permit = context.budget.inner.admission.acquire(scope).await?; + let exclusive = super::ownership::reserve(&context.target, token)?; + let tasks = context.budget.inner.tasks.clone(); + tasks + .spawn(async move { + let _permit = permit; + context + .authority + .check(&context.target, token.owner) + .await?; + let lease = context + .client + .query::(&context.target, None, ServingCheck { token, actor }) + .await + .map_err(|error| ServingReadError::Query(Box::new(error)))? + .output + .ok_or(ServingReadError::Inactive)?; + context + .authority + .check(&context.target, token.owner) + .await?; + if lease.token != token || lease.format != context.indexes.sources().format() { + return Err(ServingReadError::Context); + } + Ok(Self { + inner: Arc::new(Inner { + _exclusive: exclusive, + context, + lease, + state: Mutex::new(Workers::default()), + changed: Notify::new(), + reader: tokio::sync::Mutex::new(None), + refs: tokio::sync::OnceCell::new(), + release: tokio::sync::Mutex::new(None), + }), + }) + }) + .await? + } + pub fn token(&self) -> ServingToken { + self.inner.lease.token + } + pub fn fact(&self) -> GenerationFact { + self.inner.lease.fact + } + /// Cancellation only detaches observation. The tracked worker retains read + /// admission and the physical-drain guard until all metadata work finishes. + pub async fn headers( + &self, + actor: Option, + ids: &[crate::ObjectId], + ) -> Result>, ServingReadError> { + if ids.is_empty() + || ids.len() > PAGE_OBJECTS + || ids + .iter() + .any(|oid| oid.is_zero() || oid.format() != self.inner.lease.format) + { + return Err(ServingReadError::Context); + } + let ids = ids.to_vec(); + self.read_owned(actor, move |inner, deadline, _permit| async move { + let reader = inner.catalog().await?; + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + Ok(reader + .headers(&ids, &*inner.context.files, &*inner.context.files) + .await?) + }) + .await + } + /// Own the exact renewal and its drain guard before yielding to a caller. + /// The coordinator retains both across held/unknown states and transport loss. + pub async fn ready_renew( + &self, + actor: String, + request_digest: [u8; 32], + identity: MutationIdentity, + lease_ms: u64, + ) -> Result { + if self.inner.context.budget.inner.stop.is_cancelled() { + return Err(ServingReadError::Inactive); + } + let guard = { + let mut state = self.inner.state.lock().expect("serving workers"); + if state.closed { + return Err(ServingReadError::Inactive); + } + state.active += 1; + Arc::new(Active(Arc::clone(&self.inner))) + }; + let ctx = &self.inner.context; + ctx.authority.check(&ctx.target, self.token().owner).await?; + ReadyServingCommand::renew( + ctx.client.clone(), + ctx.target.clone(), + RenewServingRequest { + check: ServingCheck { + token: self.token(), + actor: Some(actor), + }, + lease_ms, + }, + request_digest, + identity, + guard, + ) + .await + .map_err(|error| ServingReadError::Custody(Box::new(error))) + } + /// Closing is sticky. Cancellation cannot reopen acquisition while workers + /// or a retained original release command remain owned by this service. + pub async fn close_and_drain(&self) { + self.inner.state.lock().expect("serving workers").closed = true; + loop { + let changed = self.inner.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if self.inner.state.lock().expect("serving workers").active == 0 { + return; + } + changed.await; + } + } + pub async fn ready_release( + &self, + identity: MutationIdentity, + ) -> Result { + self.close_and_drain().await; + let mut retained = self.inner.release.lock().await; + if self.inner.state.lock().expect("serving workers").released { + return Err(ServingReadError::Inactive); + } + if let Some(original) = retained.as_ref() { + return Ok(ReadyServingRelease { + inner: Arc::clone(&self.inner), + command: original.command.clone(), + digest: original.digest, + }); + } + let ctx = &self.inner.context; + ctx.authority.check(&ctx.target, self.token().owner).await?; + let sql = SqlCell::::new(ctx.client.clone(), ctx.target.clone())?; + let seed = sql + .query( + None, + super::super::sql::statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1 AND owner=?1", + vec![SqlValue::Text(ctx.administrator.clone())], + ), + ) + .await + .map_err(|error| ServingReadError::Proof(Box::new(error)))?; + let Some([seed]) = super::super::sql::rows(&seed.output)? + .first() + .map(Vec::as_slice) + else { + return Err(ServingReadError::Inactive); + }; + let data = DrainData { + tenant: *ctx.target.tenant().as_bytes(), + application: *ctx.target.application().as_bytes(), + token: self.token(), + administrator: ctx.administrator.clone(), + }; + let proof = ServingDrainProof(super::super::certificate::CertificateEnvelope::seal( + &data, + &super::super::sql::fixed(seed)?, + )?); + let mut bytes = BoundedEncoder::new(1024)?; + proof.encode(&mut bytes)?; + let digest = *blake3::hash(&bytes.finish()).as_bytes(); + let command = ctx + .client + .prepare_command::(&ctx.target, identity, proof) + .await + .map_err(|error| ServingReadError::Prepare(Box::new(error)))?; + let command = Arc::new(command); + *retained = Some(ReleaseCommand { + command: command.clone(), + digest, + }); + Ok(ReadyServingRelease { + inner: Arc::clone(&self.inner), + command, + digest, + }) + } +} +impl Inner { + fn child(self: &Arc) -> Arc { + let mut state = self.state.lock().expect("serving workers"); + state.active += 1; + Arc::new(Active(Arc::clone(self))) + } + async fn catalog(&self) -> Result, ServingReadError> { + let mut reader = self.reader.lock().await; + if reader.is_none() { + *reader = Some(Arc::new( + CatalogReader::open( + Arc::clone(&self.context.indexes), + self.lease.fact.catalog.ok_or(ServingReadError::Context)?, + ) + .await?, + )); + } + Ok(Arc::clone(reader.as_ref().expect("opened serving catalog"))) + } + async fn observe(&self, actor: Option) -> Result<(Receipt, Instant), ServingReadError> { + let ctx = &self.context; + ctx.authority + .check(&ctx.target, self.lease.token.owner) + .await?; + let started = Instant::now(); + let observed = ctx + .client + .query::( + &ctx.target, + None, + ServingCheck { + token: self.lease.token, + actor, + }, + ) + .await + .map_err(|error| ServingReadError::Query(Box::new(error)))?; + let lease = observed.output.ok_or(ServingReadError::Inactive)?; + if lease.token != self.lease.token + || lease.fact != self.lease.fact + || lease.format != self.lease.format + { + return Err(ServingReadError::Context); + } + let remaining = u64::try_from(lease.expires_at_ms - lease.observed_at_ms) + .map_err(|_| ServingReadError::Inactive)?; + let deadline = started + .checked_add(std::time::Duration::from_millis( + remaining.min(MAX_LEASE_MS), + )) + .ok_or(ServingReadError::Context)?; + ctx.authority + .check(&ctx.target, self.lease.token.owner) + .await?; + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + Ok((observed.receipt, deadline)) + } +} +#[must_use] +pub struct ReadyServingRelease { + inner: Arc, + command: Arc>, + digest: [u8; 32], +} +impl ReadyServingRelease { + pub(in crate::packs::publication) fn token(&self) -> ServingToken { + self.inner.lease.token + } + + pub fn evidence(&self) -> &cellule_runtime::PendingMutation { + self.command.evidence() + } + pub(in crate::packs::publication) fn dispatch_copy(&self) -> Self { + Self { + inner: Arc::clone(&self.inner), + command: self.command.clone(), + digest: self.digest, + } + } + pub(in crate::packs::publication) fn context( + &self, + ) -> (&CellClient, &CellTarget, BeginRequest) { + let ctx = &self.inner.context; + ( + &ctx.client, + &ctx.target, + BeginRequest { + repository: self.inner.lease.token.repository, + operation: self.inner.lease.token.reader, + request_digest: self.digest, + actor: ctx.administrator.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + } + pub(in crate::packs::publication) fn pending(&self) -> PublicationError { + PublicationError::ServingRelease(InvocationError::Pending(Box::new( + self.command.evidence().clone(), + ))) + } + pub(in crate::packs::publication) async fn dispatch( + self, + recover: bool, + fault: u8, + ) -> Result, InvocationError> { + let client = self.inner.context.client.clone(); + let inner = Arc::clone(&self.inner); + let result = super::super::exact::invoke_guarded( + &client, + (*self.command).clone(), + recover, + 128, + fault, + move || { + let state = self.inner.state.lock().expect("serving workers"); + if !state.closed || state.active != 0 { + return Err(Error::Command("serving workers have not drained")); + } + Ok(()) + }, + ) + .await; + if !matches!( + &result, + Err(InvocationError::Pending(_) | InvocationError::InvalidPublishedResult { .. }) + ) { + // Drop the cached original before the coordinator releases credits. + let mut retained = inner.release.lock().await; + if matches!(&result,Ok(value) if value.output==ServingReleaseReply::Released) { + inner.state.lock().expect("serving workers").released = true; + } + retained.take(); + } + result + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/session/body.rs b/crates/canopy-server/src/packs/publication/serving/session/body.rs new file mode 100644 index 00000000..20d253f0 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session/body.rs @@ -0,0 +1,41 @@ +//! Certified object bodies with worker ownership through native physical drain. +use super::*; + +/// Foreground copies are bounded independently of any object header. Streaming +/// larger objects is a separate producer contract, never an unbounded Vec. +pub(super) const MAX_BODY_BYTES: usize = 64 << 20; +impl ServingPin { + pub async fn body( + &self, + actor: Option, + oid: crate::ObjectId, + limit: usize, + ) -> Result>, ServingReadError> { + if oid.is_zero() || oid.format() != self.inner.lease.format { + return Err(ServingReadError::Context); + } + if limit == 0 || limit > MAX_BODY_BYTES { + return Err(ServingReadError::TooLarge); + } + self.read_owned(actor, move |inner, deadline, permit| async move { + let reader = inner.catalog().await?; + let Some(object) = reader + .lookup(oid, &*inner.context.files, &*inner.context.files) + .await? + else { + return Ok(None); + }; + if object.entry.header.object.size > limit as u64 { + return Err(ServingReadError::TooLarge); + } + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + // This is a child of an already admitted worker. Closing refuses new + // workers but must not invalidate native drain ownership of this one. + let owner: crate::git_objects::ReadOwner = Arc::new((inner.child(), permit)); + Ok(Some(inner.context.files.body(object, limit, owner).await?)) + }) + .await + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/session/edges.rs b/crates/canopy-server/src/packs/publication/serving/session/edges.rs new file mode 100644 index 00000000..1821bada --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session/edges.rs @@ -0,0 +1,85 @@ +//! Bounded typed graph pages from preferred certified metadata, never legacy SQL. +use super::*; +use crate::packs::metadata::TypedEdge; + +pub const MAX_EDGE_PARENTS: usize = 128; +#[derive(Debug, PartialEq, Eq)] +pub struct ServingEdgePage { + pub headers: Vec<(crate::ObjectId, Option)>, + pub edges: Vec<(crate::ObjectId, TypedEdge)>, + /// Conservative continuation: an exact-full page may need one empty read. + pub next_after: Option<(crate::ObjectId, crate::ObjectId)>, +} +impl ServingPin { + pub async fn edges_page( + &self, + actor: Option, + ids: &[crate::ObjectId], + after: Option<(crate::ObjectId, crate::ObjectId)>, + ) -> Result { + if ids.is_empty() + || ids.len() > MAX_EDGE_PARENTS + || ids + .iter() + .any(|oid| oid.is_zero() || oid.format() != self.inner.lease.format) + || ids.windows(2).any(|pair| pair[0] >= pair[1]) + || after.is_some_and(|(parent, child)| { + ids.binary_search(&parent).is_err() + || child.is_zero() + || child.format() != self.inner.lease.format + }) + { + return Err(ServingReadError::Context); + } + let ids = ids.to_vec(); + self.read_owned(actor, move |inner, deadline, permit| async move { + let reader = inner.catalog().await?; + let mut output = ServingEdgePage { + headers: Vec::new(), + edges: Vec::new(), + next_after: None, + }; + for parent in ids { + if after.is_some_and(|(cursor, _)| parent < cursor) { + continue; + } + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + let Some(object) = reader + .lookup(parent, &*inner.context.files, &*inner.context.files) + .await? + else { + output.headers.push((parent, None)); + continue; + }; + output.headers.push((parent, Some(object.entry.header))); + let metadata = object.source.metadata; + let cursor = after + .filter(|(cursor, _)| *cursor == parent) + .map(|(_, child)| child); + let owner = (inner.child(), permit.clone()); + let mut edges = tokio::task::spawn_blocking(move || { + let _owner = owner; + metadata.edges_after(parent, cursor) + }) + .await? + .map_err(crate::packs::directory::index::IndexError::from)?; + let keep = edges.len().min(PAGE_OBJECTS - output.edges.len()); + edges.truncate(keep); + output + .edges + .extend(edges.into_iter().map(|edge| (parent, edge))); + if output.edges.len() == PAGE_OBJECTS { + output.next_after = output + .edges + .last() + .map(|(parent, edge)| (*parent, edge.child)); + break; + } + } + Ok(output) + }) + .await + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/session/handoff.rs b/crates/canopy-server/src/packs/publication/serving/session/handoff.rs new file mode 100644 index 00000000..2df8ef8e --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session/handoff.rs @@ -0,0 +1,68 @@ +//! Accepted original knowledge retains a root; fresh observations authorize I/O. +use super::*; +use crate::packs::publication::{custody::OwnedCustody, sql}; + +impl ServingPin { + pub(in crate::packs::publication::serving) async fn retain_original( + context: ServingContext, + original: Arc, + ) -> Result { + // Cleanup remains possible after read admission closes. The trusted + // administrator's ordinary bounded node/account slot owns this probe. + let permit = context + .budget + .inner + .admission + .acquire(ReadIdentity::Account(&context.administrator)) + .await?; + let tasks = context.budget.inner.tasks.clone(); + tasks.spawn(async move { + let _permit = permit; + if original.evidence().target() != &context.target { + return Err(ServingReadError::Context); + } + let lease = original.serving_grant(&context.client).await + .map_err(|error| ServingReadError::Custody(Box::new(error)))?; + if lease.format != context.indexes.sources().format() { + return Err(ServingReadError::Context); + } + context.authority.check(&context.target, lease.token.owner).await?; + let exclusive = super::super::ownership::reserve(&context.target, lease.token)?; + let sql = SqlCell::::new(context.client.clone(), context.target.clone())?; + let mut batch = sql::statement( + "SELECT incarnation,admission_sequence,owner_epoch,generation FROM catalog_serving_pins WHERE reader=?1", + vec![SqlValue::Blob(lease.token.reader.to_vec())], + ); + batch.statements.extend(sql::statement( + sql::GENERATION, + vec![sql::number(lease.token.generation)?], + ).statements); + let sets = sql.query(None, batch).await + .map_err(|error| ServingReadError::Proof(Box::new(error)))?; + use sql::{fixed, generation, rows, unsigned}; + let (retained, generations) = sets.output.split_first().ok_or(ServingReadError::Context)?; + let Some([incarnation, sequence, epoch, retained_generation]) = rows(std::slice::from_ref(retained))?.first().map(Vec::as_slice) else { + return Err(ServingReadError::Inactive); + }; + if fixed::<16>(incarnation)? != *lease.token.owner.incarnation.as_bytes() + || unsigned(sequence)? != lease.token.admission_sequence + || u64::from_be_bytes(fixed(epoch)?) != lease.token.owner.epoch + || unsigned(retained_generation)? != lease.token.generation + || generation(generations, lease.token.repository, lease.format)? != lease.fact + { + return Err(ServingReadError::Context); + } + context.authority.check(&context.target, lease.token.owner).await?; + Ok(Self { inner: Arc::new(Inner { + _exclusive: exclusive, + context, + lease, + state: Mutex::new(Workers::default()), + changed: Notify::new(), + reader: tokio::sync::Mutex::new(None), + refs: tokio::sync::OnceCell::new(), + release: tokio::sync::Mutex::new(None), + }) }) + }).await? + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/session/native_base.rs b/crates/canopy-server/src/packs/publication/serving/session/native_base.rs new file mode 100644 index 00000000..2969b1e7 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session/native_base.rs @@ -0,0 +1,107 @@ +//! Disposable native base for write preparation. Catalog presence supplies +//! inputs, never fetch reachability or authority to publish mutations. +use super::workspace::{Observation, job}; +use super::*; +use crate::git_objects::ReadOwner; + +impl ServingPin { + pub(in crate::packs::publication::serving) async fn native_base( + &self, + actor: Option, + snapshot: ServingSnapshot, + ) -> Result { + self.read_session( + actor.clone(), + move |inner, deadline, permit| async move { + let mut observation = Observation { + deadline, + next: Instant::now(), + }; + let refs = inner.ref_snapshot().await?.clone(); + let cleanup: ReadOwner = Arc::new((inner.child(), snapshot)); + let owner: ReadOwner = Arc::new((cleanup.clone(), permit)); + let cache = inner + .context + .files + .workspace(owner.clone(), cleanup.clone(), refs.default_branch.clone()) + .await?; + let limits = WorkspaceLimits::default(); + let spool = inner + .context + .files + .graph_spool( + limits.max_spool_bytes, + limits.cache_kib, + owner.clone(), + cleanup.clone(), + ) + .await + .map_err(crate::packs::directory::index::IndexError::from)?; + let reader = inner.catalog().await?; + let mut sources = reader.source_changes(None, None)?; + loop { + observation.refresh(&inner, &actor).await?; + let page = sources.page(128, 64 << 10).await?; + if page.is_empty() { + break; + } + for record in page { + observation.refresh(&inner, &actor).await?; + let native = record.native(); + native.validate(inner.context.repository(), inner.lease.format)?; + if !job(&spool, owner.clone(), move |s| s.pack_seen(native)).await? { + inner + .context + .files + .install_workspace(cache.clone(), native, owner.clone()) + .await?; + observation.refresh(&inner, &actor).await?; + job(&spool, owner.clone(), move |s| s.imported(native)).await?; + } + } + } + // Writable native results have their own pack directory. Baseline + // catalog inputs remain immutable alternates, never incoming packs. + let cache = inner + .context + .files + .write_workspace(owner.clone(), cleanup, refs.default_branch.clone(), cache) + .await?; + let mut names = inner.context.indexes.refs().cursor(refs.root, None, true)?; + let mut writer = cache + .serving_refs(owner.clone()) + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + loop { + observation.refresh(&inner, &actor).await?; + let mut page = Vec::with_capacity(crate::refs::REF_PAGE_SIZE); + for _ in 0..crate::refs::REF_PAGE_SIZE { + let Some(record) = names.next().await? else { + break; + }; + page.push((record.name().to_owned(), record.state().clone())); + } + if page.is_empty() { + break; + } + writer = writer + .append(page) + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + } + writer + .finish() + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + inner.observe(actor).await?; + Ok(crate::git_http::GitHttpBackend { + cache, + nonce_seed: None, + signers: None, + }) + }, + true, + ) + .await + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/session/reads.rs b/crates/canopy-server/src/packs/publication/serving/session/reads.rs new file mode 100644 index 00000000..d8233e13 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session/reads.rs @@ -0,0 +1,76 @@ +//! One admission and physical lifetime for private immutable read workers. +use super::*; +use std::future::Future; + +impl ServingPin { + pub(super) async fn read_owned( + &self, + actor: Option, + work: Work, + ) -> Result + where + T: Send + 'static, + F: Future> + Send, + Work: FnOnce(Arc, Instant, Arc) -> F + Send + 'static, + { + self.read_session(actor, work, false).await + } + + /// A renewable producer retains its borrow/admission and refreshes the exact + /// lease before each bounded I/O step. Renewal cannot resurrect an expired + /// pin; the final fresh observation must still prove current authority. + pub(super) async fn read_session( + &self, + actor: Option, + work: Work, + renewable: bool, + ) -> Result + where + T: Send + 'static, + F: Future> + Send, + Work: FnOnce(Arc, Instant, Arc) -> F + Send + 'static, + { + if self.inner.context.budget.inner.stop.is_cancelled() { + return Err(ServingReadError::Inactive); + } + let scope = actor + .as_deref() + .map_or(ReadIdentity::Anonymous, ReadIdentity::Account); + let permit = self + .inner + .context + .budget + .inner + .admission + .acquire(scope) + .await?; + let guard = { + let mut state = self.inner.state.lock().expect("serving workers"); + if state.closed { + return Err(ServingReadError::Inactive); + } + state.active += 1; + Active(Arc::clone(&self.inner)) + }; + let inner = Arc::clone(&self.inner); + self.inner + .context + .tasks() + .spawn(async move { + let (permit, _guard) = (Arc::new(permit), guard); + let (_, deadline) = inner.observe(actor.clone()).await?; + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + // Observer cancellation only detaches this task. Do not time out by + // dropping provider work and misreporting that its roots drained. + let output = work(inner.clone(), deadline, permit.clone()).await?; + let (_, current) = inner.observe(actor).await?; + if Instant::now() >= current || !renewable && Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + Ok(output) + }) + .await? + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/session/refs.rs b/crates/canopy-server/src/packs/publication/serving/session/refs.rs new file mode 100644 index 00000000..379f1e34 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session/refs.rs @@ -0,0 +1,160 @@ +//! Ref facts come only from the accepted joint generation's immutable root. +use super::*; +use crate::packs::ref_state::{RefNameKey, RefStateSnapshot}; +use crate::refs::{REF_PAGE_SIZE, RefExpectation, RefPage, valid_ref_name}; + +const PAGE_BYTES: usize = 512 * 1024; + +pub struct ResolvedServingRef { + pub generation: i64, + pub reference: String, + /// None means the name never existed. A retained deletion has a version. + pub state: Option, +} + +impl Inner { + pub(super) async fn ref_snapshot(&self) -> Result<&RefStateSnapshot, ServingReadError> { + self.refs + .get_or_try_init(|| async { + let fact = self.lease.fact; + let snapshot = fact + .refs + .ok_or(ServingReadError::Context)? + .read(&self.context.indexes.store()) + .await?; + if snapshot.repository != self.lease.token.repository + || snapshot.format != self.lease.format + || snapshot.generation > fact.generation + { + return Err(ServingReadError::Context); + } + Ok(snapshot) + }) + .await + } +} + +impl ServingPin { + pub(in crate::packs::publication::serving) async fn resolve_refs( + &self, + actor: Option, + names: &[String], + ) -> Result, ServingReadError> { + if names.len() > 128 + || names.iter().any(|name| !valid_ref_name(name)) + || names.iter().map(String::len).sum::() > PAGE_BYTES + || names.windows(2).any(|p| p[0] >= p[1]) + { + return Err(ServingReadError::Context); + } + let names = names + .iter() + .map(|name| RefNameKey::new(name)) + .collect::, _>>()?; + self.read_owned(actor, move |inner, deadline, _permit| async move { + let snapshot = inner.ref_snapshot().await?; + let mut result = Vec::with_capacity(names.len()); + for name in names { + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + result.push(ResolvedServingRef { + generation: snapshot.generation as i64, + state: inner + .context + .indexes + .refs() + .read(snapshot.root.clone(), name.as_str()) + .await?, + reference: name.as_str().to_owned(), + }); + } + Ok(result) + }) + .await + } + pub async fn resolve_ref( + &self, + actor: Option, + reference: Option<&str>, + ) -> Result { + if reference.is_some_and(|name| !valid_ref_name(name)) { + return Err(ServingReadError::Context); + } + let reference = reference.map(RefNameKey::new).transpose()?; + self.read_owned(actor, move |inner, deadline, _permit| async move { + let snapshot = inner.ref_snapshot().await?; + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + let reference = reference + .as_ref() + .map_or(snapshot.default_branch.as_str(), RefNameKey::as_str); + let state = inner + .context + .indexes + .refs() + .read(snapshot.root.clone(), reference) + .await?; + Ok(ResolvedServingRef { + generation: snapshot.generation as i64, + reference: reference.to_owned(), + state, + }) + }) + .await + } + + /// Count/byte-bounded page. Live cursors skip whole deleted subtrees; other + /// consumers can retain tombstone versions without a second representation. + pub async fn refs_page( + &self, + actor: Option, + after: &str, + generation: Option, + live_only: bool, + ) -> Result { + if (!after.is_empty() && (!valid_ref_name(after) || generation.is_none())) + || generation.is_some_and(|value| value < 0) + { + return Err(ServingReadError::Context); + } + let after = (!after.is_empty()) + .then(|| RefNameKey::new(after)) + .transpose()?; + self.read_owned(actor, move |inner, deadline, _permit| async move { + let snapshot = inner.ref_snapshot().await?; + if Instant::now() >= deadline { + return Err(ServingReadError::Inactive); + } + if generation.is_some_and(|value| value != snapshot.generation as i64) { + return Err(ServingReadError::Changed); + } + let mut cursor = + inner + .context + .indexes + .refs() + .cursor(snapshot.root.clone(), after, live_only)?; + let mut refs = Vec::with_capacity(REF_PAGE_SIZE); + let mut bytes = 0; + let mut has_more = false; + while let Some(record) = cursor.next().await? { + let charge = record.name().len() + 64; + if refs.len() == REF_PAGE_SIZE || charge > PAGE_BYTES - bytes { + has_more = true; + break; + } + bytes += charge; + refs.push((record.name().to_owned(), record.state().clone())); + } + Ok(RefPage { + generation: snapshot.generation as i64, + default_branch: snapshot.default_branch.clone(), + refs, + has_more, + }) + }) + .await + } +} diff --git a/crates/canopy-server/src/packs/publication/serving/session/workspace.rs b/crates/canopy-server/src/packs/publication/serving/session/workspace.rs new file mode 100644 index 00000000..ab672874 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/serving/session/workspace.rs @@ -0,0 +1,391 @@ +//! Complete certified forward graph in admitted disk, with owned native inputs. +//! The cache may physically contain extra pack objects. Only the retained +//! membership spool authorizes wants; file presence never grants reachability. +use super::*; +use crate::packs::{catalog::graph_spool::GraphSpool, metadata::MetadataError}; +use crate::{ + ObjectId, + git_cache::GitCache, + git_objects::{GitObjects, ReadOwner}, +}; + +#[derive(Clone, Copy, Debug)] +pub struct WorkspaceLimits { + pub max_spool_bytes: u64, + pub cache_kib: u32, +} +impl Default for WorkspaceLimits { + fn default() -> Self { + Self { + max_spool_bytes: 1 << 30, + cache_kib: 256, + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct WorkspaceStats { + pub objects: u64, + pub packs: u64, + pub input_bytes: u64, +} +#[derive(Clone)] +pub struct NativeWorkspace { + core: Arc, +} +struct Core { + cache: Arc, + spool: Arc>, + pin: ServingPin, + actor: Option, + stats: WorkspaceStats, +} +impl NativeWorkspace { + pub(crate) fn backend(&self, nonce_seed: Option<[u8; 32]>) -> crate::git_http::GitHttpBackend { + crate::git_http::GitHttpBackend { + cache: self.core.cache.clone(), + nonce_seed, + signers: None, + } + } + pub fn object_format(&self) -> crate::ObjectFormat { + self.core.pin.inner.lease.format + } + pub fn fact(&self) -> GenerationFact { + self.core.pin.fact() + } + pub fn stats(&self) -> WorkspaceStats { + self.core.stats + } + // Raw paths/owners stay crate-private. A decoded DTO cannot mint a native + // read capability; producers must authorize requests and use contains. + pub(crate) fn git_dir(&self) -> std::path::PathBuf { + self.core.cache.git_dir() + } + pub(crate) fn read_owner(&self) -> ReadOwner { + self.core.clone() + } + /// Snapshot-specific forward membership, in caller order, including blobs + /// and trees. Cache presence, future refs and unrelated histories are ignored. + pub async fn contains(&self, ids: &[ObjectId]) -> Result, ServingReadError> { + if ids.len() > PAGE_OBJECTS + || ids + .iter() + .any(|id| id.is_zero() || id.format() != self.core.pin.inner.lease.format) + { + return Err(ServingReadError::Context); + } + let ids = ids.to_vec(); + let core = self.core.clone(); + self.core + .pin + .read_owned( + self.core.actor.clone(), + move |inner, _, permit| async move { + let owner = (core.clone(), inner.child(), permit); + tokio::task::spawn_blocking(move || { + let _owner = owner; + core.spool + .lock() + .map_err(|_| MetadataError::Integrity)? + .contains(&ids) + }) + .await? + .map_err(|error| crate::packs::directory::index::IndexError::from(error).into()) + }, + ) + .await + } + /// Read only objects in the completed forward closure, even when downloaded + /// packs contain other certified objects. Verify every returned native body. + pub async fn body( + &self, + oid: ObjectId, + limit: usize, + ) -> Result>, ServingReadError> { + if oid.is_zero() || oid.format() != self.core.pin.inner.lease.format { + return Err(ServingReadError::Context); + } + if limit == 0 || limit > super::body::MAX_BODY_BYTES { + return Err(ServingReadError::TooLarge); + } + let core = self.core.clone(); + let workspace = self.clone(); + self.core + .pin + .read_owned( + self.core.actor.clone(), + move |inner, _, permit| async move { + let owner: ReadOwner = + Arc::new((inner.child(), permit, workspace.read_owner())); + if !job(&core.spool, owner.clone(), move |s| s.contains(&[oid])).await?[0] { + return Ok(None); + } + let reader = inner.catalog().await?; + let object = reader + .lookup(oid, &*inner.context.files, &*inner.context.files) + .await? + .ok_or(ServingReadError::Context)?; + let expected = object.entry.header.object; + if expected.size > limit as u64 { + return Err(ServingReadError::TooLarge); + } + let mut objects = + GitObjects::batch_owned(&workspace.git_dir(), &core.cache.native, owner) + .map_err(crate::packs::catalog::NativeReadError::from)?; + let body = objects + .read_verified(expected, limit) + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + objects + .finish() + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + Ok(Some(body)) + }, + ) + .await + } +} +impl ServingPin { + pub(in crate::packs::publication::serving) async fn workspace( + &self, + actor: Option, + roots: Option<&[ObjectId]>, + limits: WorkspaceLimits, + snapshot: ServingSnapshot, + ) -> Result { + if roots.is_some_and(|roots| { + roots.is_empty() + || roots.len() > MAX_EDGE_PARENTS + || roots.windows(2).any(|p| p[0] >= p[1]) + || roots + .iter() + .any(|id| id.is_zero() || id.format() != self.inner.lease.format) + }) { + return Err(ServingReadError::Context); + } + if limits.max_spool_bytes < 16 << 10 + || limits.max_spool_bytes > canopy_object_storage::external::MAX_ARTIFACT_BYTES + || !limits.max_spool_bytes.is_multiple_of(4096) + || limits.cache_kib == 0 + || limits.cache_kib > 256 + { + return Err(ServingReadError::Context); + } + let roots = roots.map(<[ObjectId]>::to_vec); + let pin = self.clone(); + self.read_session( + actor.clone(), + move |inner, deadline, permit| async move { + let mut observation = Observation { + deadline, + next: Instant::now(), + }; + // The borrow sustains producer renewal during long construction and + // through returned native workers, descendants and cache cleanup. + let refs = if roots.is_none() { + Some(inner.ref_snapshot().await?.clone()) + } else { + None + }; + let head = refs + .as_ref() + .map_or("refs/heads/main", |refs| refs.default_branch.as_str()) + .to_owned(); + let cleanup: ReadOwner = Arc::new((inner.child(), snapshot)); + let owner: ReadOwner = Arc::new((cleanup.clone(), permit)); + let cache = inner + .context + .files + .workspace(owner.clone(), cleanup.clone(), head) + .await?; + let spool = inner + .context + .files + .graph_spool( + limits.max_spool_bytes, + limits.cache_kib, + owner.clone(), + cleanup, + ) + .await + .map_err(crate::packs::directory::index::IndexError::from)?; + if let Some(roots) = roots { + job(&spool, owner.clone(), move |s| { + s.add(&roots.into_iter().map(|id| (id, None)).collect::>()) + }) + .await?; + } else { + let refs = refs.ok_or(ServingReadError::Context)?; + let mut cursor = + inner + .context + .indexes + .refs() + .cursor(refs.root.clone(), None, true)?; + let mut writer = cache + .serving_refs(owner.clone()) + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + loop { + observation.refresh(&inner, &actor).await?; + let mut page = Vec::with_capacity(crate::refs::REF_PAGE_SIZE); + for _ in 0..crate::refs::REF_PAGE_SIZE { + let Some(record) = cursor.next().await? else { + break; + }; + page.push((record.name().to_owned(), record.state().clone())); + } + if page.is_empty() { + break; + } + let roots = page + .iter() + .map(|(_, state)| state.oid.map(|id| (id, None))) + .collect::>>() + .ok_or(ServingReadError::Context)?; + job(&spool, owner.clone(), move |s| s.add(&roots)).await?; + writer = writer + .append(page) + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + } + writer + .finish() + .await + .map_err(crate::packs::catalog::NativeReadError::from)?; + } + let reader = inner.catalog().await?; + let mut stats = WorkspaceStats { + objects: 0, + packs: 0, + input_bytes: 0, + }; + loop { + observation.refresh(&inner, &actor).await?; + let pending = job(&spool, owner.clone(), |s| s.pending()).await?; + if pending.is_empty() { + break; + } + for (id, expected) in &pending { + observation.refresh(&inner, &actor).await?; + let object = reader + .lookup(*id, &*inner.context.files, &*inner.context.files) + .await? + .ok_or(ServingReadError::Context)?; + let kind = object.entry.header.object.kind; + if expected.is_some_and(|expected| expected != kind) { + return Err(ServingReadError::Context); + } + let id = *id; + job(&spool, owner.clone(), move |s| s.add(&[(id, Some(kind))])).await?; + let source = object.source.record.native(); + source.validate(inner.context.repository(), inner.lease.format)?; + if !job(&spool, owner.clone(), move |s| s.pack_seen(source)).await? { + inner + .context + .files + .install_workspace(cache.clone(), source, owner.clone()) + .await?; + observation.refresh(&inner, &actor).await?; + job(&spool, owner.clone(), move |s| s.imported(source)).await?; + stats.packs = stats + .packs + .checked_add(1) + .ok_or(ServingReadError::TooLarge)?; + stats.input_bytes = stats + .input_bytes + .checked_add(source.pack.size) + .and_then(|n| n.checked_add(source.index.size)) + .ok_or(ServingReadError::TooLarge)?; + } + let metadata = object.source.metadata; + let mut cursor = None; + loop { + observation.refresh(&inner, &actor).await?; + let metadata = metadata.clone(); + let keep = owner.clone(); + let edges = tokio::task::spawn_blocking(move || { + let _owner = keep; + metadata.edges_after(id, cursor) + }) + .await? + .map_err(crate::packs::directory::index::IndexError::from)?; + if edges.is_empty() { + break; + } + cursor = edges.last().map(|edge| edge.child); + let count = edges.len(); + job(&spool, owner.clone(), move |s| { + s.add( + &edges + .into_iter() + .map(|edge| (edge.child, Some(edge.expected_kind))) + .collect::>(), + ) + }) + .await?; + if count < PAGE_OBJECTS { + break; + } + } + } + let count = pending.len() as u64; + job(&spool, owner.clone(), move |s| s.done(&pending)).await?; + stats.objects = stats + .objects + .checked_add(count) + .ok_or(ServingReadError::TooLarge)?; + } + inner.observe(actor.clone()).await?; + Ok(NativeWorkspace { + core: Arc::new(Core { + cache, + spool, + pin, + actor, + stats, + }), + }) + }, + true, + ) + .await + } +} +pub(super) async fn job( + spool: &Arc>, + owner: ReadOwner, + body: impl FnOnce(&mut GraphSpool) -> Result + Send + 'static, +) -> Result { + let spool = spool.clone(); + tokio::task::spawn_blocking(move || { + let _owner = owner; + let mut spool = spool.lock().map_err(|_| MetadataError::Integrity)?; + body(&mut spool) + }) + .await? + .map_err(|error| crate::packs::directory::index::IndexError::from(error).into()) +} + +/// Amortize authority queries across bounded graph steps, rather than issuing +/// repository SQL per object. Long provider suspensions still force a fresh check +/// before the next step, and construction always rechecks before returning. +pub(super) struct Observation { + pub(super) deadline: Instant, + pub(super) next: Instant, +} +impl Observation { + pub(super) async fn refresh( + &mut self, + inner: &Inner, + actor: &Option, + ) -> Result<(), ServingReadError> { + let now = Instant::now(); + if now >= self.next || now >= self.deadline { + self.deadline = inner.observe(actor.clone()).await?.1; + self.next = (Instant::now() + std::time::Duration::from_millis(250)).min(self.deadline); + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/publication/session.rs b/crates/canopy-server/src/packs/publication/session.rs index 31db7887..df62d76e 100644 --- a/crates/canopy-server/src/packs/publication/session.rs +++ b/crates/canopy-server/src/packs/publication/session.rs @@ -1,6 +1,6 @@ //! Shared authoritative preparation lease; no artifact loads or scratch. use super::*; -use cellule_runtime::{CellClient, CellTarget, MutationIdentity, Receipt}; +use cellule_runtime::{CellClient, CellTarget, Receipt}; use std::{ sync::{ Arc, Mutex, @@ -12,6 +12,7 @@ use tokio::time::Instant; #[derive(Clone)] pub struct PreparationSession { + pub(super) authority: PreparationAuthority, pub(super) client: CellClient, pub(super) target: CellTarget, pub(super) check: LeaseCheck, @@ -19,6 +20,7 @@ pub struct PreparationSession { pub(super) deadline: Arc>, pub(super) ceiling: Option, pub(super) fenced: Arc, + fence_changed: tokio::sync::watch::Sender, } impl PreparationSession { pub async fn open( @@ -26,6 +28,7 @@ impl PreparationSession { target: CellTarget, check: LeaseCheck, minimum: Option, + authority: PreparationAuthority, ) -> Result { if crate::repository_target( target.tenant(), @@ -34,11 +37,13 @@ impl PreparationSession { ) .map_err(|_| PreparationBaseError::Context)? != target + || !authority.matches(&target) { return Err(PreparationBaseError::Context); } - let (lease, deadline) = probe(&client, &target, &check, minimum).await?; + let (lease, deadline) = probe(&client, &target, &check, minimum, &authority).await?; Ok(Self { + authority, client, target, check, @@ -46,6 +51,7 @@ impl PreparationSession { deadline: Arc::new(Mutex::new(deadline)), ceiling: None, fenced: Arc::new(AtomicBool::new(false)), + fence_changed: tokio::sync::watch::channel(false).0, }) } pub(super) fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { @@ -62,45 +68,29 @@ impl PreparationSession { } Ok((self.lease, deadline)) } - /// A recorded renewal result is never a fresh clock observation. Query - /// after the durability gate even when the command is exact-outcome replay. - pub async fn renew( - &self, - identity: MutationIdentity, - lease_ms: u64, - ) -> Result<(), PreparationBaseError> { - let result = self.renew_inner(identity, lease_ms).await; + pub(super) async fn check_owner(&self) -> Result<(), PreparationBaseError> { + let result = self + .authority + .check(&self.target, self.check.token.owner) + .await; if result.is_err() { - self.fenced.store(true, Ordering::Release); + self.fence(); } result } - async fn renew_inner( - &self, - identity: MutationIdentity, - lease_ms: u64, - ) -> Result<(), PreparationBaseError> { - if self.fenced.load(Ordering::Acquire) - || self.ceiling.is_some_and(|limit| Instant::now() >= limit) - { - return Err(PreparationBaseError::Inactive); - } - let committed = self - .client - .command::( - &self.target, - identity, - LeaseRequest { - check: self.check.clone(), - lease_ms, - }, - ) - .await - .map_err(|error| PreparationBaseError::Command(Box::new(error)))?; - self.refresh(committed.receipt).await - } pub(super) fn fence(&self) { self.fenced.store(true, Ordering::Release); + // Retain the terminal value even with no observers. A worker subscribing + // after a fence must not wait for another notification. + self.fence_changed.send_replace(true); + } + pub(super) async fn wait_fenced(&self) { + let mut changed = self.fence_changed.subscribe(); + while !*changed.borrow_and_update() { + if changed.changed().await.is_err() { + return; + } + } } pub(super) async fn refresh(&self, minimum: Receipt) -> Result<(), PreparationBaseError> { let result = self.refresh_inner(minimum).await; @@ -115,8 +105,14 @@ impl PreparationSession { { return Err(PreparationBaseError::Inactive); } - let (lease, deadline) = - probe(&self.client, &self.target, &self.check, Some(minimum)).await?; + let (lease, deadline) = probe( + &self.client, + &self.target, + &self.check, + Some(minimum), + &self.authority, + ) + .await?; if lease.token != self.lease.token || lease.base != self.lease.base || lease.format != self.lease.format @@ -141,10 +137,12 @@ async fn probe( target: &CellTarget, check: &LeaseCheck, minimum: Option, + authority: &PreparationAuthority, ) -> Result<(PreparationLease, Instant), PreparationBaseError> { // Start before the query, not after its reply, so transport/queue time can // only shorten the usable lease. Queries do not replay stored commands. let started = Instant::now(); + authority.check(target, check.token.owner).await?; let lease = client .query::(target, minimum, check.clone()) .await @@ -164,5 +162,9 @@ async fn probe( if Instant::now() >= deadline { return Err(PreparationBaseError::Inactive); } + authority.check(target, check.token.owner).await?; + if Instant::now() >= deadline { + return Err(PreparationBaseError::Inactive); + } Ok((lease, deadline)) } diff --git a/crates/canopy-server/src/packs/publication/staging.rs b/crates/canopy-server/src/packs/publication/staging.rs index ba210fb0..1637b92f 100644 --- a/crates/canopy-server/src/packs/publication/staging.rs +++ b/crates/canopy-server/src/packs/publication/staging.rs @@ -219,7 +219,7 @@ pub struct ClaimStaging; impl Command for ClaimStaging { const MODULE: &'static str = RepositoryModule::NAME; const ID: u32 = 28; - const CODEC_VERSION: u32 = 1; + const CODEC_VERSION: u32 = 2; type Input = LeaseRequest; type Output = StagingReply; fn execute( @@ -237,7 +237,9 @@ impl Command for ClaimStaging { return Ok(denied(PreparationDenial::Unauthorized)); }; let Some(row) = load(context, check.token)? else { - if !super::staging_receipt::restart_matches(context, &check)? { + if !super::staging_receipt::restart_matches(context, &check)? + && !super::custody::restart_matches(context, &check, true)? + { return Ok(denied(PreparationDenial::Missing)); } let begin = BeginRequest { diff --git a/crates/canopy-server/src/packs/publication/staging_receipt.rs b/crates/canopy-server/src/packs/publication/staging_receipt.rs index 6921f563..7c57a791 100644 --- a/crates/canopy-server/src/packs/publication/staging_receipt.rs +++ b/crates/canopy-server/src/packs/publication/staging_receipt.rs @@ -1,51 +1,39 @@ //! Original initial-admission knowledge is independent of live input custody. -//! The bounded first receipt shares the logical push row and receipt codec. +pub use super::admission_receipt::AdmissionReceiptError as StagingReceiptError; use super::{ - certificate::CertificateEnvelope, - recovery::{Stamp, phase::Recorded}, - sql::*, + admission_receipt::{Admission, InitialAdmission}, + recovery::phase::Recorded, *, }; -use cellule_runtime::{ - CellClient, CellTarget, Committed, InvocationError, MutationIdentity, PendingMutation, - primitives::sql::SqlCell, -}; +use cellule_runtime::{CellClient, CellTarget, MutationIdentity}; -const DOMAIN: &[u8] = b"canopy.initial-staging-receipt.v1\0"; -#[derive(Debug, thiserror::Error)] -pub enum StagingReceiptError { - #[error("initial staging receipt binding differs")] - Context, - #[error("initial staging receipt encoding failed")] - Codec(#[from] CodecError), - #[error("initial staging receipt capability failed")] - Capability(#[from] Error), - #[error("initial staging receipt query failed")] - Query(#[source] Box>>), -} -#[derive(Clone, Debug, PartialEq, Eq)] -struct Record { - tenant: [u8; 16], - application: [u8; 16], - request: BeginRequest, - stamp: Stamp, - result: Recorded, -} -impl Record { - fn lease(&self) -> Result { - let StagingReply::Granted(lease) = self.result.decode_reply()? else { +#[derive(Clone)] +pub(super) struct Staging; +impl Admission for Staging { + const DOMAIN: &'static [u8] = b"canopy.initial-staging-receipt.v1\0"; + const COLUMN: &'static str = "initial_staging"; + type Lease = StagingLease; + type Reply = StagingReply; + fn grant(lease: StagingLease) -> StagingReply { + StagingReply::Granted(Box::new(lease)) + } + fn token(lease: &StagingLease) -> PreparationToken { + lease.token + } + fn lease(result: &Recorded, request: &BeginRequest) -> Result { + let StagingReply::Granted(lease) = result.decode_reply()? else { return Err(CodecError::Invalid( "initial staging receipt is not a grant", )); }; - if self.result.rejected() - || lease.token.repository != self.request.repository - || lease.token.operation != self.request.operation - || lease.token.request_digest != self.request.request_digest - || lease.token.attempt != self.result.sequence() + if result.rejected() + || lease.token.repository != request.repository + || lease.token.operation != request.operation + || lease.token.request_digest != request.request_digest + || lease.token.attempt != result.sequence() || lease.observed_at_ms < 0 || lease.expires_at_ms <= lease.observed_at_ms - || lease.expires_at_ms - lease.observed_at_ms != self.request.lease_ms as i64 + || lease.expires_at_ms - lease.observed_at_ms != request.lease_ms as i64 { return Err(CodecError::Invalid( "initial staging receipt result differs", @@ -54,96 +42,25 @@ impl Record { Ok(*lease) } } -impl WireValue for Record { - fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { - self.lease()?; - e.write_bytes(DOMAIN)?; - e.write_bytes(&self.tenant)?; - e.write_bytes(&self.application)?; - self.request.encode(e)?; - self.stamp.encode(e)?; - self.result.encode(e) - } - fn decode(d: &mut BoundedDecoder<'_>) -> Result { - if d.read_bytes()? != DOMAIN { - return Err(CodecError::Invalid("initial staging receipt purpose")); - } - let value = Self { - tenant: crate::packs::directory::index::codec::fixed(d)?, - application: crate::packs::directory::index::codec::fixed(d)?, - request: BeginRequest::decode(d)?, - stamp: Stamp::decode(d)?, - result: Recorded::decode(d)?, - }; - value.lease()?; - Ok(value) - } -} /// Trusted service knowledge of the first accepted Begin. It grants no upload, /// write or response permission. A restarted caller must explicitly Claim. #[derive(Clone)] -pub struct StagingAdmission { - target: CellTarget, - record: Record, -} +pub struct StagingAdmission(InitialAdmission); impl StagingAdmission { pub async fn load( client: &CellClient, target: &CellTarget, operation: [u8; 16], ) -> Result, StagingReceiptError> { - let sql = SqlCell::::new(client.clone(), target.clone())?; - let result = sql.query(None, SqlBatch { statements: vec![ - SqlStatement { sql: "SELECT actor,request_digest,initial_staging FROM pushes WHERE id=?1".into(), parameters: vec![blob(operation)] }, - SqlStatement { sql: "SELECT push_cert_seed FROM repository_identity WHERE singleton=1".into(), parameters: vec![] }, - ] }).await.map_err(|e| StagingReceiptError::Query(Box::new(e)))?; - let Some([SqlValue::Text(actor), digest, value]) = - rows(&result.output)?.first().map(Vec::as_slice) - else { - return Ok(None); - }; - let bytes = match value { - SqlValue::Null => return Ok(None), - SqlValue::Blob(bytes) => bytes, - _ => return Err(StagingReceiptError::Context), - }; - let mut d = BoundedDecoder::new(bytes, CERTIFICATE_BYTES)?; - let envelope = CertificateEnvelope::decode(&mut d)?; - d.finish()?; - let seed = attestation::seed(result.output.get(1..).ok_or(StagingReceiptError::Context)?)?; - if !envelope.authenticated(&seed) { - return Err(StagingReceiptError::Context); - } - let record: Record = envelope.data()?; - if record.request.actor != *actor - || record.request.request_digest != fixed::<32>(digest)? - || record.request.operation != operation - || record.tenant != *target.tenant().as_bytes() - || record.application != *target.application().as_bytes() - || crate::repository_target( - target.tenant(), - target.application(), - record.request.repository, - )? != *target - { - return Err(StagingReceiptError::Context); - } - Ok(Some(Self { - target: target.clone(), - record, - })) + Ok(InitialAdmission::load(client, target, operation) + .await? + .map(Self)) } pub fn lease(&self) -> StagingLease { - self.record - .lease() - .expect("authenticated initial staging receipt") + self.0.lease() } pub fn receipt(&self) -> cellule_runtime::Receipt { - cellule_runtime::Receipt { - cell: self.target.cell_id(), - incarnation: self.lease().token.owner.incarnation, - commit_sequence: self.record.result.sequence(), - } + self.0.receipt() } pub async fn ready_claim( &self, @@ -153,11 +70,11 @@ impl StagingAdmission { ) -> Result { ReadyStaging::claim( client, - self.target.clone(), + self.0.target().clone(), LeaseRequest { check: LeaseCheck { token: self.lease().token, - actor: self.record.request.actor.clone(), + actor: self.0.request().actor.clone(), }, lease_ms, }, @@ -165,116 +82,17 @@ impl StagingAdmission { ) .await } - pub(super) fn original( - &self, - evidence: &PendingMutation, - ) -> Result>, StagingReceiptError> { - if self.record.stamp != Stamp::of(evidence) { - return Ok(None); - } - if evidence.target() != &self.target - || evidence.incarnation() != self.lease().token.owner.incarnation - { - return Err(StagingReceiptError::Context); - } - Ok(Some(self.record.result.committed(evidence)?)) - } } -/// Called after admission writes. An encoding/SQL failure aborts admission and -/// its SDK receipt together. This never replaces an earlier logical receipt. pub(super) fn save( context: &CommandContext<'_, '_>, request: &BeginRequest, lease: StagingLease, ) -> cellule_runtime::Result<()> { - let evidence = context - .mutation_evidence() - .ok_or(Error::Command("initial staging evidence missing"))?; - let mut output = BoundedEncoder::new(512)?; - StagingReply::Granted(Box::new(lease)).encode(&mut output)?; - let record = Record { - tenant: *evidence.target().tenant().as_bytes(), - application: *evidence.target().application().as_bytes(), - request: request.clone(), - stamp: Stamp::of(&evidence), - result: Recorded::new(context.sequence(), false, output.finish())?, - }; - let seed = attestation::seed(&context.sql(&statement( - "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", - vec![], - ))?)?; - let envelope = CertificateEnvelope::seal(&record, &seed)?; - let mut encoded = BoundedEncoder::new(CERTIFICATE_BYTES)?; - envelope.encode(&mut encoded)?; - let existing = context.sql(&statement( - "SELECT actor,request_digest,initial_staging FROM pushes WHERE id=?1", - vec![blob(request.operation)], - ))?; - let saved = if let Some(row) = rows(&existing)?.first() { - let [SqlValue::Text(actor), digest, original] = row.as_slice() else { - return Err(Error::Command("invalid initial staging push row")); - }; - if *actor != request.actor || fixed::<32>(digest)? != request.request_digest { - return Err(Error::Command("initial staging push identity differs")); - } - if *original != SqlValue::Null { - return Ok(()); - } - statement( - "UPDATE pushes SET initial_staging=?1 WHERE id=?2 AND initial_staging IS NULL AND response_id IS NULL", - vec![blob(encoded.finish()), blob(request.operation)], - ) - } else { - statement( - "INSERT INTO pushes(id,actor,request_digest,initial_staging) VALUES(?1,?2,?3,?4)", - vec![ - blob(request.operation), - SqlValue::Text(request.actor.clone()), - blob(request.request_digest), - blob(encoded.finish()), - ], - ) - }; - publish::changed(context.sql(&saved)?)?; - Ok(()) + super::admission_receipt::save::(context, request, lease) } - -/// Only original authenticated admission knowledge can recreate a missing -/// staging operation. Completed logical outcomes still refuse recreation. pub(super) fn restart_matches( context: &CommandContext<'_, '_>, check: &LeaseCheck, ) -> cellule_runtime::Result { - let sets = context.sql(&statement( - "SELECT actor,request_digest,initial_staging FROM pushes WHERE id=?1", - vec![blob(check.token.operation)], - ))?; - let Some([SqlValue::Text(actor), digest, SqlValue::Blob(bytes)]) = - rows(&sets)?.first().map(Vec::as_slice) - else { - return Ok(false); - }; - if *actor != check.actor || fixed::<32>(digest)? != check.token.request_digest { - return Ok(false); - } - let seed = attestation::seed(&context.sql(&statement( - "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", - vec![], - ))?)?; - let mut d = BoundedDecoder::new(bytes, CERTIFICATE_BYTES)?; - let envelope = CertificateEnvelope::decode(&mut d)?; - d.finish()?; - if !envelope.authenticated(&seed) { - return Err(Error::Command( - "initial staging receipt authentication failed", - )); - } - let record: Record = envelope.data()?; - let evidence = context - .mutation_evidence() - .ok_or(Error::Command("staging Claim evidence missing"))?; - Ok(record.request.actor == check.actor - && record.lease()?.token == check.token - && record.tenant == *evidence.target().tenant().as_bytes() - && record.application == *evidence.target().application().as_bytes()) + super::admission_receipt::restart_matches::(context, check) } diff --git a/crates/canopy-server/src/packs/publication/staging_service.rs b/crates/canopy-server/src/packs/publication/staging_service.rs index 890a92e6..011d135a 100644 --- a/crates/canopy-server/src/packs/publication/staging_service.rs +++ b/crates/canopy-server/src/packs/publication/staging_service.rs @@ -1,5 +1,6 @@ //! Service-owned, bounded input custody. Never infer a live deadline from a //! replayed command, discard ambiguous evidence, or let an observer cancel work. +use super::custody::OwnedCustody; use super::*; use crate::packs::catalog::{CatalogFiles, CatalogIndexes}; use cellule_runtime::{ @@ -20,6 +21,8 @@ use tokio::{ mod bound; use bound::accept_bound; mod publication; +mod restore; +mod retirement; pub use publication::{StagedPublicationFailure, StagedPublicationTicket}; const COMMAND_BYTES: u32 = 4096; @@ -108,15 +111,17 @@ pub enum StagingError { BoundRenew(#[source] Box>), #[error("bound preparation claim failed")] BoundClaim(#[source] Box>), + #[error("restored staging custody command failed")] + Restoration(#[source] Box>), #[error("staging bind failed")] Bind(#[source] Box>), - #[error("initial staging receipt recovery failed")] - ReceiptRecovery { + #[error("staging custody preparation failed")] + CustodyReady(#[source] Box), + #[error("staging custody protocol failed")] + Custody { evidence: Box, - source: Box, + source: Box, }, - #[error("initial staging receipt lookup failed")] - ReceiptLookup(#[source] Box), #[error("staging query failed")] Query(#[source] Box>>), #[error("bound base failed")] @@ -126,9 +131,9 @@ pub enum StagingError { #[error("final publication failed: {0}")] Publication(#[source] Arc), } -impl From for StagingError { - fn from(error: StagingReceiptError) -> Self { - Self::ReceiptLookup(Box::new(error)) +impl From for StagingError { + fn from(error: CustodyError) -> Self { + Self::CustodyReady(Box::new(error)) } } impl StagingError { @@ -142,7 +147,8 @@ impl StagingError { match self { Self::Begin(e) | Self::Renew(e) | Self::Claim(e) | Self::Checkpoint(e) => unknown(e), Self::Bind(e) | Self::BoundRenew(e) | Self::BoundClaim(e) => unknown(e), - Self::ReceiptRecovery { .. } => true, + Self::Restoration(e) => unknown(e), + Self::Custody { source, .. } => source.uncertain(), _ => false, } } @@ -204,23 +210,25 @@ impl ReadyStaging { request .encode(&mut BoundedEncoder::new(COMMAND_BYTES).map_err(|_| StagingError::Context)?) .map_err(|_| StagingError::Context)?; - if StagingAdmission::load(&client, &target, request.operation) + if RegisteredCustody::load_latest(&client, &target, request.operation) .await? .is_some() { return Err(StagingError::Duplicate); } - let command = client - .prepare_command::(&target, identity, request.clone()) - .await - .map_err(|e| StagingError::Begin(Box::new(e)))?; - let operation = request.operation; + let command = OwnedCustody::prepare( + &client, + &target, + CustodyAction::BeginStaging(request.clone()), + identity, + ) + .await?; Ok(Self { inner: Box::new(StagingRequest { client, target, request, - command: Exact::Begin(command, operation), + command: Exact::Begin(command), bound_source: None, }), }) @@ -249,10 +257,13 @@ impl ReadyStaging { request .encode(&mut BoundedEncoder::new(COMMAND_BYTES).map_err(|_| StagingError::Context)?) .map_err(|_| StagingError::Context)?; - let command = client - .prepare_command::(&target, identity, request) - .await - .map_err(|e| StagingError::Claim(Box::new(e)))?; + let command = OwnedCustody::prepare( + &client, + &target, + CustodyAction::ClaimStaging(request), + identity, + ) + .await?; Ok(Self { inner: Box::new(StagingRequest { client, @@ -288,10 +299,13 @@ impl ReadyStaging { .encode(&mut BoundedEncoder::new(COMMAND_BYTES).map_err(|_| StagingError::Context)?) .map_err(|_| StagingError::Context)?; let source = request.check.clone(); - let command = client - .prepare_command::(&target, identity, request) - .await - .map_err(|e| StagingError::BoundClaim(Box::new(e)))?; + let command = OwnedCustody::prepare( + &client, + &target, + CustodyAction::ClaimPreparation(request), + identity, + ) + .await?; Ok(Self { inner: Box::new(StagingRequest { client, @@ -310,10 +324,16 @@ struct ActorAdmission { #[derive(Default)] struct Admission { closed: bool, + retirement_probe: bool, + retirement_probes: u64, + retirement_failures: u64, + retirement_recoveries: u64, + retirement_restarts: u64, jobs: HashMap<[u8; 16], Arc>, actors: HashMap, } struct Inner { + authority: PreparationAuthority, target: CellTarget, limits: StagingLimits, admission: Mutex, @@ -329,6 +349,7 @@ struct Local { bound_started: Option, bound_result: Option>, bound_renewal: Option>, + restored_outcome: Option>>, policy_receipt: Option, finishing: bool, deadline: Instant, @@ -350,10 +371,12 @@ struct WorkSlots { slots: HashMap>, } struct Job { + authority: PreparationAuthority, client: CellClient, target: CellTarget, actor: String, operation: [u8; 16], + restored_evidence: Option, actor_workers: Arc, local: Mutex, work: Mutex, @@ -365,16 +388,18 @@ struct Job { } #[derive(Clone)] enum Exact { - Begin(PreparedCommand, [u8; 16]), - Claim(PreparedCommand), + Restored(OwnedCustody), + Begin(OwnedCustody), + Claim(OwnedCustody), Checkpoint(PreparedCommand), BoundCheckpoint(PreparedCommand), - Renew(PreparedCommand), - Bind(PreparedCommand), - BoundClaim(PreparedCommand), - BoundRenew(PreparedCommand), + Renew(OwnedCustody), + Bind(OwnedCustody), + BoundClaim(OwnedCustody), + BoundRenew(OwnedCustody), } enum Outcome { + Restored(Box>), Stage(Committed), Bound(Committed), BoundClaim(Committed), @@ -382,10 +407,40 @@ enum Outcome { Checkpoint(Committed), BoundCheckpoint(Committed), } +fn custody_guard(job: &Job) -> Result<(), Error> { + let local = job + .local + .lock() + .map_err(|_| Error::Command("staging custody poisoned"))?; + let now = Instant::now(); + if local.fenced + || now >= local.lifetime + || ((local.lease.is_some() || local.bound.is_some()) && now >= local.deadline) + { + return Err(Error::Command("staging custody inactive")); + } + Ok(()) +} + impl Exact { + fn custody_original(&self) -> Option<&OwnedCustody> { + match self { + Self::Restored(c) + | Self::Begin(c) + | Self::Claim(c) + | Self::Renew(c) + | Self::Bind(c) + | Self::BoundClaim(c) + | Self::BoundRenew(c) => Some(c), + Self::Checkpoint(_) | Self::BoundCheckpoint(_) => None, + } + } fn pending(&self) -> StagingError { match self { - Self::Begin(c, _) => StagingError::Begin(Box::new(InvocationError::Pending(Box::new( + Self::Restored(c) => StagingError::Restoration(Box::new(InvocationError::Pending( + Box::new(c.evidence().clone()), + ))), + Self::Begin(c) => StagingError::Begin(Box::new(InvocationError::Pending(Box::new( c.evidence().clone(), )))), Self::Claim(c) => StagingError::Claim(Box::new(InvocationError::Pending(Box::new( @@ -408,41 +463,93 @@ impl Exact { )))), } } + async fn custody( + command: OwnedCustody, + client: &CellClient, + recover: bool, + fault: u8, + job: &Job, + project: fn(CustodyReply) -> Option, + role: fn(Box>) -> StagingError, + ) -> Result, StagingError> { + let result = command + .invoke(client, recover, fault, || custody_guard(job)) + .await + .map_err(|source| StagingError::Custody { + evidence: Box::new(command.evidence().clone()), + source: Box::new(source), + })?; + super::custody::project(result, project).map_err(|error| role(Box::new(error))) + } async fn execute( self, client: CellClient, + job: Arc, recover: bool, fault: u8, ) -> Result { + fn stage(reply: CustodyReply) -> Option { + match reply { + CustodyReply::Staging(reply) => Some(reply), + _ => None, + } + } + fn preparation(reply: CustodyReply) -> Option { + match reply { + CustodyReply::Preparation(reply) => Some(reply), + _ => None, + } + } match self { - Self::Begin(c, operation) => { - if recover { - let known = async { - match StagingAdmission::load(&client, c.evidence().target(), operation) - .await? - { - Some(saved) => saved.original(c.evidence()), - None => Ok(None), - } - } + Self::Restored(c) => restore::dispatch(c, &client, &job, recover, fault).await, + Self::Begin(c) => { + Self::custody(c, &client, recover, fault, &job, stage, StagingError::Begin) .await - .map_err(|source| StagingError::ReceiptRecovery { - evidence: Box::new(c.evidence().clone()), - source: Box::new(source), - })?; - if let Some(value) = known { - return Ok(Outcome::Stage(value)); - } - } - super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .map(Outcome::Stage) + } + Self::Claim(c) => { + Self::custody(c, &client, recover, fault, &job, stage, StagingError::Claim) .await .map(Outcome::Stage) - .map_err(|e| StagingError::Begin(Box::new(e))) } - Self::Claim(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) - .await - .map(Outcome::Stage) - .map_err(|e| StagingError::Claim(Box::new(e))), + Self::Renew(c) => { + Self::custody(c, &client, recover, fault, &job, stage, StagingError::Renew) + .await + .map(Outcome::Stage) + } + Self::Bind(c) => Self::custody( + c, + &client, + recover, + fault, + &job, + preparation, + StagingError::Bind, + ) + .await + .map(Outcome::Bound), + Self::BoundClaim(c) => Self::custody( + c, + &client, + recover, + fault, + &job, + preparation, + StagingError::BoundClaim, + ) + .await + .map(Outcome::BoundClaim), + Self::BoundRenew(c) => Self::custody( + c, + &client, + recover, + fault, + &job, + preparation, + StagingError::BoundRenew, + ) + .await + .map(Outcome::BoundRenew), Self::Checkpoint(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) .await .map(Outcome::Checkpoint) @@ -453,22 +560,6 @@ impl Exact { .map(Outcome::BoundCheckpoint) .map_err(|e| StagingError::Checkpoint(Box::new(e))) } - Self::Renew(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) - .await - .map(Outcome::Stage) - .map_err(|e| StagingError::Renew(Box::new(e))), - Self::BoundClaim(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) - .await - .map(Outcome::BoundClaim) - .map_err(|e| StagingError::BoundClaim(Box::new(e))), - Self::BoundRenew(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) - .await - .map(Outcome::BoundRenew) - .map_err(|e| StagingError::BoundRenew(Box::new(e))), - Self::Bind(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) - .await - .map(Outcome::Bound) - .map_err(|e| StagingError::Bind(Box::new(e))), } } } @@ -493,12 +584,26 @@ pub struct StagingStats { pub uncertain: usize, pub command_bytes: u64, pub closed: bool, + /// Read-only exact-original probes, outside the command wire reservation. + pub retirement_probes: u64, + pub retirement_failures: u64, + pub retirement_recoveries: u64, + pub retirement_restarts: u64, + pub retirement_running: bool, } impl StagingCoordinator { - pub fn new(target: CellTarget, limits: StagingLimits) -> Result { + pub fn new( + target: CellTarget, + limits: StagingLimits, + authority: PreparationAuthority, + ) -> Result { limits.validate()?; + if !authority.matches(&target) { + return Err(StagingError::Context); + } Ok(Self { inner: Arc::new(Inner { + authority, target, limits, admission: Mutex::new(Admission::default()), @@ -519,7 +624,9 @@ impl StagingCoordinator { Some(StagingError::Foreign) } else if admission.closed { Some(StagingError::Closed) - } else if ready.inner.request.lease_ms != self.inner.limits.lease_ms { + } else if !matches!(ready.inner.command, Exact::Restored(_)) + && ready.inner.request.lease_ms != self.inner.limits.lease_ms + { Some(StagingError::Context) } else if admission.jobs.contains_key(&ready.inner.request.operation) { Some(StagingError::Duplicate) @@ -548,19 +655,28 @@ impl StagingCoordinator { actor.operations += 1; let actor_workers = Arc::clone(&actor.workers); let now = Instant::now(); + let restored_evidence = match &ready.inner.command { + Exact::Restored(command) => Some(command.evidence().clone()), + _ => None, + }; let job = Arc::new(Job { + authority: self.inner.authority.clone(), client: ready.inner.client, target: ready.inner.target, actor: ready.inner.request.actor, operation: ready.inner.request.operation, + restored_evidence, actor_workers, local: Mutex::new(Local { lease: None, bound: None, bound_source: ready.inner.bound_source.clone(), - bound_started: ready.inner.bound_source.as_ref().map(|_| now), + bound_started: (ready.inner.bound_source.is_some() + || matches!(ready.inner.command, Exact::Restored(_))) + .then_some(now), bound_result: None, bound_renewal: None, + restored_outcome: None, policy_receipt: None, finishing: false, deadline: now, @@ -625,12 +741,17 @@ impl StagingCoordinator { .jobs .values() .map(|job| { - let copies = - 2 + u64::from(job.checkpoint.lock().expect("staging checkpoint").is_some()); - copies * COMMAND_BYTES as u64 + super::custody::RESERVATION + + u64::from(job.checkpoint.lock().expect("staging checkpoint").is_some()) + * u64::from(COMMAND_BYTES) }) .sum(), closed: a.closed, + retirement_probes: a.retirement_probes, + retirement_failures: a.retirement_failures, + retirement_recoveries: a.retirement_recoveries, + retirement_restarts: a.retirement_restarts, + retirement_running: a.retirement_probe, } } /// Stop admission and renew while accepted workers drain. Uncertain exact @@ -722,6 +843,29 @@ impl StagedInputsTicket { } } impl StagingTicket { + #[cfg(test)] + pub(super) fn custody_evidence_for_test( + &self, + ) -> Option<( + cellule_runtime::PendingMutation, + Option, + )> { + let exact = self.job.exact.lock().expect("staging exact"); + let command = match exact.as_ref()? { + Exact::Restored(command) + | Exact::Begin(command) + | Exact::Claim(command) + | Exact::Renew(command) + | Exact::Bind(command) + | Exact::BoundClaim(command) + | Exact::BoundRenew(command) => command, + Exact::Checkpoint(_) | Exact::BoundCheckpoint(_) => return None, + }; + Some(( + command.evidence().clone(), + command.registration_evidence().cloned(), + )) + } /// Synchronously transfer one bounded checkpoint request into service /// custody. A dropped observer cannot cancel or replace its exact identity. pub fn register_inputs( @@ -893,6 +1037,20 @@ impl StagingTicket { .bound_result .clone() } + /// Original registered identity retained even when fresh custody fails. + /// This is historical evidence, never upload or preparation permission. + pub fn restored_evidence(&self) -> Option<&cellule_runtime::PendingMutation> { + self.job.restored_evidence.as_ref() + } + /// Original positive or negative outcome, retained before fresh probes. + pub fn restored_outcome(&self) -> Option>> { + self.job + .local + .lock() + .expect("staging local") + .restored_outcome + .clone() + } pub fn bound_renewal(&self) -> Option> { self.job .local @@ -916,12 +1074,12 @@ impl StagingTicket { /// slots, cancellation/drain and typed result ownership as staging inputs. pub fn spawn_bound(&self, producer: F) -> Result, StagingError> where - F: FnOnce(Arc) -> Fut + Send + 'static, + F: FnOnce(Arc, StagingContext) -> Fut + Send + 'static, Fut: Future> + Send + 'static, T: Send + 'static, { let session = self.bound_session()?; - self.spawn_in_phase(true, move |_| producer(session)) + self.spawn_in_phase(true, move |context| producer(session, context)) } fn spawn_in_phase( &self, @@ -939,7 +1097,7 @@ impl StagingTicket { let actor_permit = Arc::clone(&self.job.actor_workers) .try_acquire_owned() .map_err(|_| StagingError::Capacity)?; - let context = { + let (token, format) = { let mut l = self.job.local.lock().expect("staging local"); if (l.seal && !bound) || l.finishing @@ -966,18 +1124,20 @@ impl StagingTicket { (lease.token, lease.format) }; l.workers += 1; - StagingContext { - job: Arc::clone(&self.job), - token, - format, - bound, - } + (token, format) }; - let guard = Activity { + let guard = Arc::new(Activity { inner: Arc::clone(&self.inner), job: Arc::clone(&self.job), permit: Some(permit), actor_permit: Some(actor_permit), + }); + let context = StagingContext { + job: Arc::clone(&self.job), + token, + format, + bound, + activity: Arc::clone(&guard), }; let mut work = self.job.work.lock().expect("staging work"); let id = work.next.checked_add(1).ok_or(StagingError::Capacity)?; @@ -1050,6 +1210,77 @@ impl StagingTicket { }) } #[cfg(test)] + pub(super) async fn renew_with_identity_for_test( + &self, + identity: MutationIdentity, + ) -> Result<(), StagingError> { + let (token, bound) = { + let local = self.job.local.lock().expect("staging local"); + if local.fenced { + return Err(StagingError::Inactive); + } + match &local.bound { + Some(session) => (session.live_lease()?.0.token, true), + None => (local.lease.ok_or(StagingError::NotReady)?.token, false), + } + }; + let lease = LeaseRequest { + check: LeaseCheck { + token, + actor: self.job.actor.clone(), + }, + lease_ms: self.inner.limits.lease_ms, + }; + let action = if bound { + CustodyAction::RenewPreparation(lease) + } else { + CustodyAction::RenewStaging(lease) + }; + let command = + OwnedCustody::prepare(&self.job.client, &self.job.target, action, identity).await?; + let mut exact = self.job.exact.lock().expect("staging exact"); + assert!(exact.is_none(), "test renewal replaced an original"); + *exact = Some(if bound { + Exact::BoundRenew(command) + } else { + Exact::Renew(command) + }); + drop(exact); + self.job.changed.notify_one(); + Ok(()) + } + /// Inject a short custody ceiling only after a test reaches its intended + /// recovery phase. The real clock and normal fence/drain path still run. + #[cfg(test)] + pub(super) fn limit_bound_ceiling_for_test( + &self, + remaining: Duration, + ) -> Result { + let mut local = self.job.local.lock().expect("staging local"); + if local.workers != 0 || local.finishing || remaining.is_zero() { + return Err(StagingError::Context); + } + let mut session = (**local.bound.as_ref().ok_or(StagingError::NotReady)?).clone(); + session.live_lease()?; + let ceiling = (Instant::now() + remaining).min(local.lifetime); + session.ceiling = Some(ceiling); + local.bound = Some(Arc::new(session)); + local.lifetime = ceiling; + self.job.changed.notify_one(); + Ok(ceiling) + } + #[cfg(test)] + pub(super) fn expire_bound_for_test(&self) -> Result { + let mut local = self.job.local.lock().expect("staging local"); + let session = local.bound.as_ref().ok_or(StagingError::NotReady)?; + let deadline = (Instant::now() + Duration::from_millis(100)).min(session.live_lease()?.1); + *session.deadline.lock().expect("bound deadline") = deadline; + local.deadline = deadline; + local.lifetime = local.lifetime.min(deadline); + self.job.changed.notify_one(); + Ok(deadline) + } + #[cfg(test)] pub(super) fn renew_for_test(&self) { self.job.local.lock().expect("staging local").renew = true; self.job.changed.notify_one(); @@ -1076,8 +1307,16 @@ pub struct StagingContext { token: PreparationToken, format: ObjectFormat, bound: bool, + // Clones share the original worker admission. Detached blocking jobs, + // native descendants and provider readers must retain it until they drain. + activity: Arc, } impl StagingContext { + /// Lifetime ownership only; it grants no custody or publication authority. + /// Keep this in every physical worker that can outlive its async observer. + pub(crate) fn physical_owner(&self) -> crate::git_objects::ReadOwner { + self.activity.clone() + } pub(super) fn capability(&self) -> (&CellClient, &CellTarget, LeaseCheck) { ( &self.job.client, @@ -1099,6 +1338,9 @@ impl StagingContext { let l = self.job.local.lock().expect("staging local"); if l.fenced || l.bound.is_some() != self.bound + || l.bound + .as_ref() + .is_some_and(|session| session.live_lease().is_err()) || l.deadline <= Instant::now() || l.lifetime <= Instant::now() { @@ -1110,24 +1352,40 @@ impl StagingContext { async fn fenced(&self) { let mut status = self.job.status.subscribe(); loop { - let deadline = { + let (deadline, session) = { let l = self.job.local.lock().expect("staging local"); if l.fenced || l.bound.is_some() != self.bound { return; } - l.deadline.min(l.lifetime) + let mut deadline = l.deadline.min(l.lifetime); + if let Some(session) = &l.bound { + let Ok((_, usable_until)) = session.live_lease() else { + return; + }; + deadline = deadline.min(usable_until); + } + (deadline, l.bound.clone()) }; if Instant::now() >= deadline { return; } - tokio::select! { _ = sleep_until(deadline) => {}, result = status.changed() => { if result.is_err() { return; } } } + tokio::select! { + _ = sleep_until(deadline) => {}, + _ = async { + match session { + Some(session) => session.wait_fenced().await, + None => std::future::pending::<()>().await, + } + } => return, + result = status.changed() => { if result.is_err() { return; } } + } } } } struct WorkSlot { result: Mutex>>>, ready: watch::Sender, - guard: Mutex>, + guard: Mutex>>, } impl RetainedWork for WorkSlot { fn erased(self: Arc) -> Arc { @@ -1187,6 +1445,7 @@ async fn probe(job: &Job, minimum: Receipt) -> Result<(StagingLease, Instant), S Some(l) => l.token, None => return Err(StagingError::Context), }; + job.authority.check(&job.target, token.owner).await?; let lease = job .client .query::( @@ -1213,6 +1472,10 @@ async fn probe(job: &Job, minimum: Receipt) -> Result<(StagingLease, Instant), S if deadline <= Instant::now() { return Err(StagingError::Inactive); } + job.authority.check(&job.target, token.owner).await?; + if Instant::now() >= deadline { + return Err(StagingError::Inactive); + } Ok((lease, deadline)) } async fn supervise(inner: Arc, job: Arc) { @@ -1251,6 +1514,9 @@ async fn supervise(inner: Arc, job: Arc) { job.status .send_replace(StagingState::Uncertain(Arc::new(exact.pending()))); inner.drained.notify_waiters(); + if exact.custody_original().is_some() { + retirement::start(Arc::clone(&inner)); + } await_recovery(&job).await; recover = true; } @@ -1301,15 +1567,27 @@ async fn run(inner: Arc, job: Arc, mut recover: bool) { let fault = inner.fault.swap(0, std::sync::atomic::Ordering::AcqRel); #[cfg(not(test))] let fault = 0; + let custody = command.custody_original().is_some(); let pending = command.pending(); - let result = tokio::spawn(command.execute(job.client.clone(), recover, fault)) - .await - .unwrap_or(Err(pending)); + let result = + tokio::spawn(command.execute(job.client.clone(), Arc::clone(&job), recover, fault)) + .await + .unwrap_or(Err(pending)); match result { + Ok(Outcome::Restored(value)) => { + job.exact.lock().expect("staging exact").take(); + if !restore::accept(&inner, &job, *value).await { + return; + } + recover = false; + } Err(error) if error.uncertain() => { job.status .send_replace(StagingState::Uncertain(Arc::new(error))); inner.drained.notify_waiters(); + if custody { + retirement::start(Arc::clone(&inner)); + } await_recovery(&job).await; recover = true; continue; @@ -1410,6 +1688,7 @@ async fn run(inner: Arc, job: Arc, mut recover: bool) { | Outcome::BoundCheckpoint(_) => { unreachable!() } + Outcome::Restored(_) => unreachable!("restored outcome handled above"), }; let StagingReply::Granted(lease) = value.output else { job.exact.lock().expect("staging exact").take(); @@ -1680,12 +1959,15 @@ async fn run(inner: Arc, job: Arc, mut recover: bool) { job.status.send_replace(StagingState::Binding); let identity = crate::server::mutation_identity().map_err(|_| StagingError::Clock); let result = match identity { - Ok(id) => job - .client - .prepare_command::(&job.target, id, check) - .await - .map(Exact::Bind) - .map_err(|e| StagingError::Bind(Box::new(e))), + Ok(id) => OwnedCustody::prepare( + &job.client, + &job.target, + CustodyAction::BindStaging(check), + id, + ) + .await + .map(Exact::Bind) + .map_err(StagingError::from), Err(e) => Err(e), }; match result { @@ -1701,19 +1983,18 @@ async fn run(inner: Arc, job: Arc, mut recover: bool) { Next::BoundRenew(check) => { let identity = crate::server::mutation_identity().map_err(|_| StagingError::Clock); let result = match identity { - Ok(id) => job - .client - .prepare_command::( - &job.target, - id, - LeaseRequest { - check, - lease_ms: inner.limits.lease_ms, - }, - ) - .await - .map(Exact::BoundRenew) - .map_err(|e| StagingError::BoundRenew(Box::new(e))), + Ok(id) => OwnedCustody::prepare( + &job.client, + &job.target, + CustodyAction::RenewPreparation(LeaseRequest { + check, + lease_ms: inner.limits.lease_ms, + }), + id, + ) + .await + .map(Exact::BoundRenew) + .map_err(StagingError::from), Err(e) => Err(e), }; match result { @@ -1729,19 +2010,18 @@ async fn run(inner: Arc, job: Arc, mut recover: bool) { Next::Renew(check) => { let identity = crate::server::mutation_identity().map_err(|_| StagingError::Clock); let result = match identity { - Ok(id) => job - .client - .prepare_command::( - &job.target, - id, - LeaseRequest { - check, - lease_ms: inner.limits.lease_ms, - }, - ) - .await - .map(Exact::Renew) - .map_err(|e| StagingError::Renew(Box::new(e))), + Ok(id) => OwnedCustody::prepare( + &job.client, + &job.target, + CustodyAction::RenewStaging(LeaseRequest { + check, + lease_ms: inner.limits.lease_ms, + }), + id, + ) + .await + .map(Exact::Renew) + .map_err(StagingError::from), Err(e) => Err(e), }; match result { diff --git a/crates/canopy-server/src/packs/publication/staging_service/bound.rs b/crates/canopy-server/src/packs/publication/staging_service/bound.rs index 8aab4f1a..13d81a62 100644 --- a/crates/canopy-server/src/packs/publication/staging_service/bound.rs +++ b/crates/canopy-server/src/packs/publication/staging_service/bound.rs @@ -47,6 +47,17 @@ pub(super) async fn accept_bound( fence_and_drain(inner, job, StagingError::Context).await; return false; } + open_bound(inner, job, original, ceiling).await +} + +/// Shared fresh session installation after authenticating a bound outcome. +pub(super) async fn open_bound( + inner: &Inner, + job: &Job, + original: Arc, + ceiling: Instant, +) -> bool { + let lease = original.lease; let session = PreparationSession::open( job.client.clone(), job.target.clone(), @@ -54,7 +65,8 @@ pub(super) async fn accept_bound( token: lease.token, actor: job.actor.clone(), }, - Some(value.receipt), + Some(original.receipt), + job.authority.clone(), ) .await; let mut session = match session { diff --git a/crates/canopy-server/src/packs/publication/staging_service/restore.rs b/crates/canopy-server/src/packs/publication/staging_service/restore.rs new file mode 100644 index 00000000..8f147f65 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/staging_service/restore.rs @@ -0,0 +1,159 @@ +//! Reconstruct only an authenticated registered original, never a new retry. +use super::*; + +impl ReadyStaging { + /// Load the exact latest registered custody head after process loss. + /// No recorded clock grants work; the service resolves original knowledge + /// before independently acquiring current lease/owner custody. + pub async fn restore( + client: CellClient, + target: CellTarget, + operation: [u8; 16], + ) -> Result { + let command = OwnedCustody::restore(&client, &target, operation).await?; + let request = match command.action().map_err(|_| StagingError::Context)? { + CustodyAction::BeginPreparation(request) | CustodyAction::BeginStaging(request) => { + request + } + CustodyAction::ClaimPreparation(request) + | CustodyAction::RenewPreparation(request) + | CustodyAction::ClaimStaging(request) + | CustodyAction::RenewStaging(request) => context(request.check, request.lease_ms), + // Bind has no requested renewal duration. This synthetic value is + // admission metadata only, never execution bytes or a lease clock. + CustodyAction::BindStaging(check) => context(check, DEFAULT_LEASE_MS), + CustodyAction::AcquireServing(_) | CustodyAction::RenewServing { .. } => { + return Err(StagingError::Context); + } + }; + if request.operation != operation + || crate::repository_target(target.tenant(), target.application(), request.repository) + .map_err(|_| StagingError::Context)? + != target + { + return Err(StagingError::Context); + } + request + .encode(&mut BoundedEncoder::new(COMMAND_BYTES).map_err(|_| StagingError::Context)?) + .map_err(|_| StagingError::Context)?; + Ok(Self { + inner: Box::new(StagingRequest { + client, + target, + request, + command: Exact::Restored(command), + bound_source: None, + }), + }) + } +} +fn context(check: LeaseCheck, lease_ms: u64) -> BeginRequest { + BeginRequest { + repository: check.token.repository, + operation: check.token.operation, + request_digest: check.token.request_digest, + actor: check.actor, + lease_ms, + } +} + +pub(super) async fn dispatch( + command: OwnedCustody, + client: &CellClient, + job: &Job, + recover: bool, + fault: u8, +) -> Result { + // Only frozen command execution occurs here, not new native work. Known + // phases precede the local guard. On proven SDK absence the original + // receiver checks live custody and actual ownership atomically; successful + // execution still cannot grant a worker before the fresh post-result probe. + let result = command + .invoke(client, recover, fault, || custody_guard(job)) + .await + .map_err(|source| StagingError::Custody { + evidence: Box::new(command.evidence().clone()), + source: Box::new(source), + })?; + let value = match result { + Ok(value) => value, + Err(InvocationError::Rejected(value)) => *value, + Err(error) => return Err(StagingError::Restoration(Box::new(error))), + }; + Ok(Outcome::Restored(Box::new(value))) +} + +pub(super) async fn accept(inner: &Inner, job: &Job, value: Committed) -> bool { + // Preserve positive AND negative knowledge before any current authority, + // lease query, scope ceiling or newly configured duration can refuse work. + let value = Arc::new(value); + job.local.lock().expect("staging local").restored_outcome = Some(value.clone()); + match &value.output { + CustodyReply::Staging(StagingReply::Granted(recorded)) => { + job.local.lock().expect("staging local").lease = Some(**recorded); + match probe(job, value.receipt).await { + Ok((lease, deadline)) + if lease.token == recorded.token && lease.format == recorded.format => + { + let active = { + let mut local = job.local.lock().expect("staging local"); + if local.fenced || Instant::now() >= deadline.min(local.lifetime) { + false + } else { + local.lease = Some(lease); + local.deadline = deadline; + job.status.send_replace(if local.stop { + StagingState::Draining(lease) + } else { + StagingState::Active(lease) + }); + true + } + }; + if !active { + fence_and_drain(inner, job, StagingError::Inactive).await; + } + active + } + Ok(_) => { + fence_and_drain(inner, job, StagingError::Context).await; + false + } + Err(error) => { + fence_and_drain(inner, job, error).await; + false + } + } + } + CustodyReply::Preparation(PreparationReply::Granted(lease)) => { + let original = Arc::new(StagingBound { + lease: **lease, + receipt: value.receipt, + }); + let ceiling = { + let mut local = job.local.lock().expect("staging local"); + local.bound_result = Some(original.clone()); + let started = local + .bound_started + .expect("restored custody admission time"); + let ceiling = (started + Duration::from_millis(inner.limits.bound_lifetime_ms)) + .min(local.lifetime); + local.lifetime = ceiling; + (!local.fenced && Instant::now() < ceiling).then_some(ceiling) + }; + match ceiling { + Some(ceiling) => bound::open_bound(inner, job, original, ceiling).await, + None => { + fence_and_drain(inner, job, StagingError::Inactive).await; + false + } + } + } + CustodyReply::Serving(_) + | CustodyReply::Staging(StagingReply::Denied(_)) + | CustodyReply::Preparation(PreparationReply::Denied(_)) => { + fence_and_drain(inner, job, StagingError::Context).await; + false + } + } +} diff --git a/crates/canopy-server/src/packs/publication/staging_service/retirement.rs b/crates/canopy-server/src/packs/publication/staging_service/retirement.rs new file mode 100644 index 00000000..84c36342 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/staging_service/retirement.rs @@ -0,0 +1,150 @@ +//! One read-only sweeper per service, over existing bounded admitted jobs. +use super::*; +use std::collections::BinaryHeap; + +pub(super) fn start(inner: Arc) { + let mut admission = inner.admission.lock().expect("staging admission"); + if admission.retirement_probe { + return; + } + admission.retirement_probe = true; + drop(admission); + tokio::spawn(async move { + loop { + if tokio::spawn(run(Arc::clone(&inner))).await.is_ok() { + return; + } + { + let mut admission = inner.admission.lock().expect("staging admission"); + admission.retirement_restarts = admission.retirement_restarts.saturating_add(1); + } + tracing::error!("staging retirement probe failed; retaining exact jobs and restarting"); + tokio::time::sleep(RecoveryScanLimits::default().interval).await; + } + }); +} +fn eligible(job: &Job) -> bool { + matches!(*job.status.borrow(), StagingState::Uncertain(_)) + && job + .exact + .lock() + .expect("staging exact") + .as_ref() + .is_some_and(|exact| exact.custody_original().is_some()) +} +fn page(inner: &Inner, after: &mut Option<[u8; 16]>, limit: usize) -> Option>> { + let mut admission = inner.admission.lock().expect("staging admission"); + let mut keys = BinaryHeap::with_capacity(limit + 1); + for pass in 0..2 { + for (operation, job) in &admission.jobs { + if after.is_none_or(|cursor| *operation > cursor) && eligible(job) { + keys.push(*operation); + if keys.len() > limit { + keys.pop(); + } + } + } + if !keys.is_empty() { + break; + } + if pass == 0 { + *after = None; + } + } + if keys.is_empty() { + // The same lock covers start/exit, so a newly uncertain job cannot lose + // its wakeup between observing an empty page and relinquishing ownership. + admission.retirement_probe = false; + return None; + } + Some( + keys.into_sorted_vec() + .into_iter() + .map(|key| Arc::clone(&admission.jobs[&key])) + .collect(), + ) +} +async fn visit(inner: &Arc, job: &Arc) { + let probe = { + let exact = job.exact.lock().expect("staging exact"); + exact + .as_ref() + .and_then(Exact::custody_original) + .map(OwnedCustody::stop_probe) + }; + let Some(probe) = probe else { + return; + }; + let probe = match probe { + Ok(probe) => probe, + Err(error) => { + failed(inner, &error); + return; + } + }; + // No ready command/body is cloned across this await. This independent read + // uses SDK query admission, and never performs registration or execution. + let stopped = probe.observed(&job.client).await; + { + let mut admission = inner.admission.lock().expect("staging admission"); + admission.retirement_probes = admission.retirement_probes.saturating_add(1); + } + match stopped { + Ok(true) => {} + Ok(false) => return, + Err(error) => { + failed(inner, &error); + return; + } + } + let same_original = job + .exact + .lock() + .expect("staging exact") + .as_ref() + .and_then(Exact::custody_original) + .is_some_and(|current| current.evidence() == probe.evidence()); + if !same_original { + return; + } + let ticket = StagingTicket { + inner: Arc::clone(inner), + job: Arc::clone(job), + }; + // Exact recovery reauthenticates closure, fences the shared session and + // drains resources before returning admission. Closure never becomes a reply. + if (StagingCoordinator { + inner: Arc::clone(inner), + }) + .recover(&ticket) + .is_ok() + { + let mut admission = inner.admission.lock().expect("staging admission"); + admission.retirement_recoveries = admission.retirement_recoveries.saturating_add(1); + } +} +fn failed(inner: &Inner, error: &CustodyError) { + let mut admission = inner.admission.lock().expect("staging admission"); + admission.retirement_failures = admission.retirement_failures.saturating_add(1); + tracing::debug!(%error, "staging retirement probe unavailable; retaining original"); +} +async fn run(inner: Arc) { + let limits = RecoveryScanLimits::default(); + let mut after = None; + loop { + let Some(jobs) = page(&inner, &mut after, usize::from(limits.page)) else { + return; + }; + let deadline = Instant::now() + limits.interval; + for job in jobs { + after = Some(job.operation); + visit(&inner, &job).await; + // A slow failed head advances the cursor without consuming the rest + // of this round. Keep the read owner until its query finishes. + if Instant::now() >= deadline { + break; + } + } + tokio::time::sleep(limits.interval).await; + } +} diff --git a/crates/canopy-server/src/packs/publication/tests.rs b/crates/canopy-server/src/packs/publication/tests.rs index 723f347c..821c7ff4 100644 --- a/crates/canopy-server/src/packs/publication/tests.rs +++ b/crates/canopy-server/src/packs/publication/tests.rs @@ -3,16 +3,21 @@ mod attestation; mod compaction; mod completion; mod coordinator; +mod custody; +mod custody_stop; mod durable_policy; mod durable_recovery; mod frontier; mod initialization; +mod initialization_recovery; +mod initialization_retirement; mod inputs; mod mandatory_registration; mod namespaces; mod native_capture; mod policy_dispatch; mod policy_refusal; +mod preparation_receipt; mod prepare; mod publishing; mod reconcile; @@ -23,6 +28,7 @@ mod refs; mod root_completion; mod root_dispatch; mod root_outcome; +mod serving; mod staged_durable; mod staging; mod staging_receipt; @@ -67,37 +73,31 @@ impl CellModule for Module { publish_descriptor.input_limit = 4 << 20; let mut complete_descriptor = descriptor(19); complete_descriptor.input_limit = 4 << 20; - let mut initial_descriptor = descriptor(31); - initial_descriptor.input_limit = INITIALIZATION_BYTES; - initial_descriptor.output_limit = 512; - let mut initial_query = descriptor(32); - initial_query.output_limit = 512; - let mut policy_page = descriptor(33); - policy_page.codec_version = RegisterRefPolicyPage::CODEC_VERSION; - policy_page.input_limit = REF_POLICY_PAGE_BYTES; - policy_page.output_limit = 128; - let mut policy_query = descriptor(34); - policy_query.output_limit = 128; - let mut policy_reap = descriptor(35); - policy_reap.output_limit = 128; - let mut root_completion = descriptor(36); - root_completion.codec_version = CompleteRootPush::CODEC_VERSION; - root_completion.input_limit = ROOT_COMPLETION_BYTES; - root_completion.output_limit = 512; - let mut root_outcome = descriptor(38); - root_outcome.codec_version = CompleteRootOutcome::CODEC_VERSION; - root_outcome.input_limit = ROOT_COMPLETION_BYTES; - root_outcome.output_limit = 512; - let mut root_lookup = descriptor(37); - root_lookup.output_limit = 512; - let mut recovery = descriptor(39); - recovery.codec_version = RegisterRootRecovery::CODEC_VERSION; - // Match the existing production SQL transport contract exactly. - // The generic 4 KiB fixture limit cannot encode even one valid - // 65 KiB ref name; policy construction has its own smaller bound. - let sql_query = crate::operation(2); - let mut release = descriptor(40); - release.output_limit = 128; + let mut commands = super::registry::COMMANDS.to_vec(); + commands.extend([publish_descriptor, complete_descriptor, ref_descriptor]); + for id in [AcquireServingPin::ID, RenewServingPin::ID] { + let mut raw = descriptor(id); + raw.input_limit = 1024; + raw.output_limit = 1024; + commands.push(raw); + } + // Raw domain receivers qualify their invariants here. Production + // binds only the mandatory registered custody envelope. + for (id, codec) in [ + (BeginPreparation::ID, BeginPreparation::CODEC_VERSION), + (ClaimPreparation::ID, ClaimPreparation::CODEC_VERSION), + (RenewPreparation::ID, RenewPreparation::CODEC_VERSION), + (BeginStaging::ID, BeginStaging::CODEC_VERSION), + (ClaimStaging::ID, ClaimStaging::CODEC_VERSION), + (RenewStaging::ID, RenewStaging::CODEC_VERSION), + (BindStaging::ID, BindStaging::CODEC_VERSION), + ] { + let mut domain = descriptor(id); + domain.codec_version = codec; + commands.push(domain); + } + let mut queries = super::registry::QUERIES.to_vec(); + queries.push(descriptor(20)); ModuleDescriptor { name: Self::NAME, source_digest: Digest::from_bytes([11; 32]), @@ -109,42 +109,8 @@ impl CellModule for Module { sql: SCHEMA, digest: Digest::from_bytes(*blake3::hash(SCHEMA.as_bytes()).as_bytes()), }])), - commands: Box::leak(Box::new([ - descriptor(11), - descriptor(12), - descriptor(13), - descriptor(14), - descriptor(16), - descriptor(17), - publish_descriptor, - complete_descriptor, - descriptor(22), - descriptor(24), - descriptor(25), - descriptor(26), - descriptor(28), - descriptor(29), - initial_descriptor, - policy_page, - policy_reap, - root_completion, - root_outcome, - recovery, - release, - ref_descriptor, - ])), - queries: Box::leak(Box::new([ - sql_query, - descriptor(15), - descriptor(20), - descriptor(21), - descriptor(23), - descriptor(27), - descriptor(30), - initial_query, - policy_query, - root_lookup, - ])), + commands: Box::leak(commands.into_boxed_slice()), + queries: Box::leak(queries.into_boxed_slice()), workflow_definitions: &[], activity_types: &[], namespaces: Box::leak(Box::new([NamespaceDescriptor { @@ -159,9 +125,21 @@ impl CellModule for Module { }) } fn register(self, registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { + cellule_runtime::primitives::sql::register_sql::(registry)?; super::register(registry)?; - registry.bind_command::()?; - registry.bind_query::>() + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_query::()?; + registry.bind_command::() } } struct Fixture { @@ -174,8 +152,16 @@ struct Fixture { registry: Arc, runtime: CellRuntime, handle: CellHandle, + publication_budget: PublicationBudget, + scan_budget: RecoveryScanBudget, } impl Fixture { + fn scans(&self, limits: RecoveryScanLimits) -> RecoveryScanSettings { + self.scan_budget.settings(limits, "owner") + } + fn authority(&self) -> PreparationAuthority { + PreparationAuthority::local(self.layout.clone(), self.target.clone()) + } async fn new(format: ObjectFormat) -> Result { Self::with_artifact_sequence(format, 0).await } @@ -239,6 +225,16 @@ impl Fixture { registry, runtime, handle, + publication_budget: PublicationBudget::new(PublicationLimits { + operations: 128, + per_actor: 32, + command_bytes: 512 << 20, + in_flight: 16, + maintenance_operations: 16, + maintenance_in_flight: 4, + ..PublicationLimits::default() + })?, + scan_budget: RecoveryScanBudget::new(8, tokio_util::task::TaskTracker::new())?, }) } fn client(&self) -> CellClient { @@ -256,6 +252,18 @@ impl Fixture { async fn counts(&self) -> Result<(u64, u64)> { counts(&self.handle).await } + async fn counts_for(&self, token: PreparationToken) -> Result<(u64, u64)> { + let bytes = self.handle.query(0, 16, move |connection| { + let parameters = rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt]; + let operations: u64 = connection.query_row("SELECT count(*) FROM catalog_operations WHERE incarnation=?1 AND admission_sequence=?2", parameters, |row| row.get(0))?; + let leases: u64 = connection.query_row("SELECT count(*) FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", parameters, |row| row.get(0))?; + Ok([operations.to_be_bytes(), leases.to_be_bytes()].concat()) + }).await?; + Ok(( + u64::from_be_bytes(bytes[..8].try_into()?), + u64::from_be_bytes(bytes[8..].try_into()?), + )) + } // Trusted fixture injection only. Production generation facts require the // complete catalog verifier and fenced publisher; a digest is not a proof. async fn install_catalog(&self, generation: u64, catalog: StoredCatalog) -> Result<()> { @@ -329,6 +337,22 @@ fn identity() -> std::io::Result { expires_at_ms: now + 60_000, }) } +async fn registered_preparation( + f: &Fixture, + operation: [u8; 16], +) -> Result> { + Ok(PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginPreparation(f.begin(operation)), + identity()?, + ) + .await? + .register(&f.client(), identity()?) + .await? + .recover_preparation(&f.client()) + .await?) +} fn lease(reply: PreparationReply) -> Result { match reply { PreparationReply::Granted(lease) => Ok(*lease), @@ -1050,9 +1074,7 @@ async fn authoritative_base_resolution_uses_live_queried_facts_and_fences_failed .await?; fixture.install_catalog(1, native.stored).await?; let client = fixture.client(); - let started = client - .command::(&fixture.target, identity()?, fixture.begin([36; 16])) - .await?; + let started = registered_preparation(&fixture, [36; 16]).await?; let granted = lease(started.output)?; let budget = DiskBudget::new(128 << 20); let files = Arc::new(CatalogFiles::new( @@ -1073,6 +1095,7 @@ async fn authoritative_base_resolution_uses_live_queried_facts_and_fences_failed Arc::clone(&native.indexes), Arc::clone(&files), Some(started.receipt), + fixture.authority(), ) .await?; let base = resolver.context().base.ok_or("base")?; @@ -1099,7 +1122,20 @@ async fn authoritative_base_resolution_uses_live_queried_facts_and_fences_failed Err(ClosureError::Integrity) )); let renewal = identity()?; - resolver.renew(renewal, DEFAULT_LEASE_MS).await?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; + let ticket = coordinator + .submit(resolver.ready_renew(renewal, DEFAULT_LEASE_MS).await?) + .await?; + let PublicationState::Finished(Ok(PublicationOutcome::Preparation(outcome))) = + ticket.wait().await + else { + return Err("renewal did not finish".into()); + }; + outcome.session.map_err(|e| e.to_string())?; fixture .handle .execute( @@ -1119,10 +1155,20 @@ async fn authoritative_base_resolution_uses_live_queried_facts_and_fences_failed .await?; // The exact renewal RPC replays success, but the subsequent fresh query // sees expiry. It cannot restart a local deadline from the old reply. + let replay = coordinator + .submit(resolver.restore_renewal().await?) + .await?; + let PublicationState::Finished(Ok(PublicationOutcome::Preparation(replayed))) = + replay.wait().await + else { + return Err("renewal replay did not finish".into()); + }; + assert_eq!(replayed.committed, outcome.committed); assert!(matches!( - resolver.renew(renewal, DEFAULT_LEASE_MS).await, - Err(PreparationBaseError::Inactive) + &replayed.session, + Err(error) if matches!(&**error, PreparationBaseError::Inactive) )); + assert!(coordinator.close_and_drain().await.is_empty()); assert!(matches!( resolver.resolve(base, &ids).await, Err(ClosureError::LeaseExpired) @@ -1134,7 +1180,8 @@ async fn authoritative_base_resolution_uses_live_queried_facts_and_fences_failed check(granted.token), Arc::clone(&native.indexes), Arc::clone(&files), - None + None, + fixture.authority(), ) .await, Err(PreparationBaseError::Inactive) diff --git a/crates/canopy-server/src/packs/publication/tests/compaction/coordinator.rs b/crates/canopy-server/src/packs/publication/tests/compaction/coordinator.rs index 35c56508..49eec3fd 100644 --- a/crates/canopy-server/src/packs/publication/tests/compaction/coordinator.rs +++ b/crates/canopy-server/src/packs/publication/tests/compaction/coordinator.rs @@ -16,7 +16,11 @@ async fn maintenance_final_publication_uses_shared_bound_lifecycle_and_reserved_ let fixture = Fixture::new(format).await?; let inventory = seed(&fixture, 2).await?; let before_refs = refs(&fixture.handle).await?; - let stages = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let stages = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::new( fixture.client(), fixture.target.clone(), @@ -49,7 +53,7 @@ async fn maintenance_final_publication_uses_shared_bound_lifecycle_and_reserved_ let owned_root = root.clone(); let owned_budget = budget.clone(); let mutation = identity()?; - let work = ticket.spawn_bound(move |_| async move { + let work = ticket.spawn_bound(move |_, _context| async move { let prepared = Arc::new( PreparedCompaction::prepare( owned_root.path(), @@ -76,8 +80,11 @@ async fn maintenance_final_publication_uses_shared_bound_lifecycle_and_reserved_ Ok((ready, weak)) })?; let (ready, weak) = work.wait().await.map_err(|e| e.to_string())?; - let publications = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let publications = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let (release, entered) = publications.pause_for_test().await; let observer = ticket.publish(&publications, ready)?; timeout(Duration::from_secs(10), entered).await??; @@ -124,8 +131,11 @@ async fn uncertain_compaction_retains_exact_command_and_recovers_original_receip let compact = Arc::new(prepared.compact); let weak = Arc::downgrade(&compact); let ready = compact.ready_compaction(identity()?).await?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; coordinator.fault_for_test(fault); let ticket = coordinator.try_reserve(ready)?; assert_eq!(ticket.class(), PublicationClass::Maintenance); @@ -236,6 +246,7 @@ async fn reserved_classes_and_actor_quotas_keep_mixed_admission_bounded() -> Res maintenance_in_flight: 1, foreground_burst: 3, }, + fixture.publication_budget.clone(), )?; let (release, entered) = coordinator.pause_for_test().await; let mut entered = Some(entered); @@ -372,8 +383,11 @@ async fn queued_compaction_rechecks_admin_and_canceled_observer_cannot_cancel_pu let prepared = prepare_compaction(&fixture, &inventory, 180, &[0, 1]).await?; let compact = Arc::new(prepared.compact); let weak = Arc::downgrade(&compact); - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let (release, entered) = coordinator.pause_for_test().await; let ticket = coordinator .submit(compact.ready_compaction(identity()?).await?) @@ -424,6 +438,7 @@ async fn maintenance_concurrency_cap_keeps_foreground_progressing() -> Result { maintenance_in_flight: 1, ..PublicationLimits::default() }, + fixture.publication_budget.clone(), )?; let (release, entered) = coordinator.pause_for_test().await; let a = coordinator diff --git a/crates/canopy-server/src/packs/publication/tests/compaction/schedule.rs b/crates/canopy-server/src/packs/publication/tests/compaction/schedule.rs index 9170ffc2..9fd2565b 100644 --- a/crates/canopy-server/src/packs/publication/tests/compaction/schedule.rs +++ b/crates/canopy-server/src/packs/publication/tests/compaction/schedule.rs @@ -13,8 +13,11 @@ async fn geometric_planner_drains_native_ingress_and_level_debt_without_changing urgent_burst: 2, }; let mut planner = CompactionPlanner::new(policy)?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let mut expected = None; let mut first_catalog = None; let mut jobs = 0; diff --git a/crates/canopy-server/src/packs/publication/tests/completion.rs b/crates/canopy-server/src/packs/publication/tests/completion.rs index 1de04b12..678dc8b7 100644 --- a/crates/canopy-server/src/packs/publication/tests/completion.rs +++ b/crates/canopy-server/src/packs/publication/tests/completion.rs @@ -143,7 +143,13 @@ pub(super) async fn counts(handle: &CellHandle) -> Result> { ] .into_iter() .map(|table| { - connection.query_row(&format!("SELECT count(*) FROM {table}"), [], |row| { + // Admission-only rows are retained knowledge, not completed + // push/publication outcomes. Other counts still include every + // response/certificate/chunk written by the final transaction. + let sql = if table == "pushes" { + "SELECT count(*) FROM pushes WHERE response_id IS NOT NULL OR publication IS NOT NULL".into() + } else { format!("SELECT count(*) FROM {table}") }; + connection.query_row(&sql, [], |row| { row.get::<_, u64>(0) }) }) diff --git a/crates/canopy-server/src/packs/publication/tests/completion/outcome.rs b/crates/canopy-server/src/packs/publication/tests/completion/outcome.rs index dd946a1d..6d43c646 100644 --- a/crates/canopy-server/src/packs/publication/tests/completion/outcome.rs +++ b/crates/canopy-server/src/packs/publication/tests/completion/outcome.rs @@ -26,6 +26,7 @@ async fn opened(fixture: &Fixture, operation: [u8; 16]) -> Result Result { limits(), )) .await?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let (release, entered) = coordinator.pause_for_test().await; let ticket = coordinator.submit(ready).await?; entered.await?; @@ -550,6 +582,7 @@ async fn bounded_dispatch_allows_another_command_to_progress_before_first_outcom maintenance_in_flight: 1, ..PublicationLimits::default() }, + fixture.publication_budget.clone(), )?; let (release, entered) = coordinator.pause_for_test().await; let ready = Box::pin(first.0.ready_push( @@ -621,8 +654,11 @@ async fn oversized_inline_completion_fails_before_dispatch_while_inventory_stays limits(), )) .await?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let ticket = coordinator.submit(ready).await?; finished(ticket.wait().await)?; assert!(coordinator.close_and_drain().await.is_empty()); diff --git a/crates/canopy-server/src/packs/publication/tests/coordinator/budget.rs b/crates/canopy-server/src/packs/publication/tests/coordinator/budget.rs new file mode 100644 index 00000000..e3b3fb32 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/coordinator/budget.rs @@ -0,0 +1,225 @@ +use super::*; + +fn node_limits() -> PublicationLimits { + PublicationLimits { + operations: 8, + per_actor: 2, + command_bytes: (24 << 20) + (32 << 10), + in_flight: 4, + maintenance_operations: 4, + maintenance_in_flight: 2, + foreground_burst: 3, + } +} + +async fn prepared_outcome( + fixture: &Fixture, + operation: u8, + actor: &str, +) -> Result<( + ReadyCatalogPush, + tempfile::TempDir, + DiskBudget, + std::sync::Weak, +)> { + let (prepared, root, disk) = empty(fixture, [operation; 16], actor).await?; + let session = Arc::new(prepared.base.session.clone()); + let weak = Arc::downgrade(&session); + let ready = session + .ready_outcome(identity()?, request(refused())) + .await?; + drop((prepared, session)); + cleaned(root.path(), &disk).await?; + Ok((ready, root, disk, weak)) +} + +#[tokio::test] +async fn held_commands_charge_one_node_across_repositories_and_return_exact_refusals() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let first = Fixture::new(format).await?; + let second = Fixture::new(format).await?; + for fixture in [&first, &second] { + edit( + fixture, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + } + let node = PublicationBudget::new(node_limits())?; + let a = PublicationCoordinator::new( + first.target.clone(), + PublicationLimits::default(), + node.clone(), + )?; + let b = PublicationCoordinator::new( + second.target.clone(), + PublicationLimits::default(), + node.clone(), + )?; + let (ready, root1, disk1, weak1) = prepared_outcome(&first, 51, "owner").await?; + let t1 = a.try_reserve(ready)?; + let (ready, root2, disk2, weak2) = prepared_outcome(&second, 52, "owner").await?; + let t2 = b.try_reserve(ready)?; + let (ready, root3, disk3, weak3) = prepared_outcome(&second, 53, "owner").await?; + let original = ready.evidence_for_test(); + let failure = b + .try_reserve(ready) + .err() + .ok_or("aggregate account quota bypassed")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + let ReadyPublication::Push(ready) = failure.ready else { + return Err("ready variant changed".into()); + }; + assert_eq!(ready.evidence_for_test(), original); + let (writer, root4, disk4, weak4) = prepared_outcome(&second, 54, "writer").await?; + let t4 = b.try_reserve(writer)?; + let (writer, root5, disk5, weak5) = prepared_outcome(&first, 55, "writer").await?; + let writer_original = writer.evidence_for_test(); + let failure = a + .try_reserve(writer) + .err() + .ok_or("aggregate command bytes bypassed")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + let ReadyPublication::Push(writer) = failure.ready else { + return Err("ready variant changed".into()); + }; + assert_eq!(writer.evidence_for_test(), writer_original); + let stats = node.stats(); + assert_eq!( + ( + stats.foreground, + stats.command_bytes, + stats.accounts, + stats.foreground_dispatch + ), + (3, 24 << 20, 2, 0) + ); + assert!(weak1.upgrade().is_some()); + // Observer handles survive credit return, but the retained session + // must already be gone when that return is observed. + t1.discard_held().await?; + assert!(weak1.upgrade().is_none()); + cleaned(root1.path(), &disk1).await?; + assert_eq!(node.stats().foreground, 2); + let t3 = b.try_reserve(ready)?; + assert_eq!(node.stats().foreground, 3); + t2.discard_held().await?; + let t5 = a.try_reserve(writer)?; + node.close(); + let (ready, root6, disk6, weak6) = prepared_outcome(&first, 56, "writer").await?; + let original = ready.evidence_for_test(); + let failure = a + .try_reserve(ready) + .err() + .ok_or("closed node admitted new work")?; + assert_eq!(failure.reason, PublicationScheduleError::Closed); + let ReadyPublication::Push(ready) = failure.ready else { + return Err("ready variant changed".into()); + }; + assert_eq!(ready.evidence_for_test(), original); + drop(ready); + assert!(weak6.upgrade().is_none()); + for ticket in [&t3, &t4, &t5] { + ticket.discard_held().await?; + } + assert!(a.close_and_drain().await.is_empty()); + assert!(b.close_and_drain().await.is_empty()); + assert!(matches!(t1.state(), PublicationState::Discarded)); + assert_eq!( + ( + node.stats().foreground, + node.stats().accounts, + node.stats().command_bytes + ), + (0, 0, 0) + ); + for (root, disk, weak) in [ + (root2, disk2, weak2), + (root3, disk3, weak3), + (root4, disk4, weak4), + (root5, disk5, weak5), + (root6, disk6, weak6), + ] { + assert!(weak.upgrade().is_none()); + cleaned(root.path(), &disk).await?; + } + first.runtime.shutdown().await?; + second.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn account_transport_across_repositories_does_not_block_another_account() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let first = Fixture::new(format).await?; + let second = Fixture::new(format).await?; + edit( + &second, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + let node = PublicationBudget::new(node_limits())?; + let a = PublicationCoordinator::new( + first.target.clone(), + PublicationLimits::default(), + node.clone(), + )?; + let b = PublicationCoordinator::new( + second.target.clone(), + PublicationLimits::default(), + node.clone(), + )?; + let (release, entered) = a.pause_for_test().await; + let (ready, root1, disk1, weak1) = prepared_outcome(&first, 61, "owner").await?; + let t1 = a.submit(ready).await?; + timeout(Duration::from_secs(5), entered).await??; + let (ready, root2, disk2, weak2) = prepared_outcome(&second, 62, "owner").await?; + let t2 = b.submit(ready).await?; + assert!(timeout(Duration::from_millis(30), t2.wait()).await.is_err()); + assert_eq!( + (node.stats().foreground, node.stats().foreground_dispatch), + (2, 1) + ); + drop(t2); + assert!(weak2.upgrade().is_some()); + let (ready, root3, disk3, weak3) = prepared_outcome(&second, 63, "writer").await?; + let writer = b.submit(ready).await?; + finished(timeout(Duration::from_secs(10), writer.wait()).await?)?; + assert_eq!(writer.response().await?, refused()); + assert!(weak3.upgrade().is_none()); + assert_eq!( + (node.stats().foreground, node.stats().foreground_dispatch), + (2, 1) + ); + node.close(); + let t2 = b.pending([62; 16]).await.ok_or("waiting original lost")?; + release.send(()).map_err(|_| "paused dispatcher lost")?; + for ticket in [&t1, &t2] { + finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert_eq!(ticket.response().await?, refused()); + } + assert!(a.close_and_drain().await.is_empty()); + assert!(b.close_and_drain().await.is_empty()); + assert_eq!( + ( + node.stats().foreground, + node.stats().foreground_dispatch, + node.stats().command_bytes, + node.stats().accounts + ), + (0, 0, 0, 0) + ); + for (root, disk, weak) in [ + (root1, disk1, weak1), + (root2, disk2, weak2), + (root3, disk3, weak3), + ] { + assert!(weak.upgrade().is_none()); + cleaned(root.path(), &disk).await?; + } + first.runtime.shutdown().await?; + second.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/coordinator/held.rs b/crates/canopy-server/src/packs/publication/tests/coordinator/held.rs index 8b116b76..922f9576 100644 --- a/crates/canopy-server/src/packs/publication/tests/coordinator/held.rs +++ b/crates/canopy-server/src/packs/publication/tests/coordinator/held.rs @@ -20,8 +20,11 @@ async fn held_native_proof_survives_canceled_observation_and_closed_activation() limits(), )) .await?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let ticket = coordinator.try_reserve(ready)?; drop(prepared); let observer = ticket.clone(); @@ -102,14 +105,22 @@ async fn held_discard_drops_native_proof_before_credit_and_never_executes() -> R limits(), )) .await?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let ticket = coordinator.try_reserve(ready)?; drop(prepared); assert_eq!(coordinator.close_and_drain().await.len(), 1); ticket.discard_held().await?; assert!(matches!(ticket.wait().await, PublicationState::Discarded)); assert!(weak.upgrade().is_none()); + let node = fixture.publication_budget.stats(); + assert_eq!( + (node.foreground, node.command_bytes, node.accounts), + (0, 0, 0) + ); cleaned(graph.root.path(), &graph.budget).await?; assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); assert!(coordinator.pending([60; 16]).await.is_none()); @@ -152,6 +163,7 @@ async fn held_admission_uses_existing_account_bytes_and_returns_refused_ready() maintenance_in_flight: 1, foreground_burst: 3, }, + fixture.publication_budget.clone(), )?; let mut attempts = Vec::new(); for (n, actor) in [ @@ -179,6 +191,7 @@ async fn held_admission_uses_existing_account_bytes_and_returns_refused_ready() fixture.repository, )?, PublicationLimits::default(), + fixture.publication_budget.clone(), )?; let failure = foreign .try_reserve(attempts[0].3.take().unwrap()) @@ -244,8 +257,11 @@ async fn held_admission_uses_existing_account_bytes_and_returns_refused_ready() .err() .ok_or("refused admission unexpectedly accepted")?; assert_eq!(failure.reason, PublicationScheduleError::Closed); - let successor = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let successor = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let ticket = successor.try_reserve(failure.ready)?; ticket.activate().await?; finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; @@ -271,8 +287,11 @@ async fn activated_held_command_recovers_exact_receipt_or_checks_absent_authorit limits(), )) .await?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let ticket = coordinator.try_reserve(ready)?; assert!(matches!(ticket.state(), PublicationState::Held)); coordinator.fault_for_test(fault); @@ -361,8 +380,11 @@ async fn activated_held_command_recovers_exact_receipt_or_checks_absent_authorit #[tokio::test] async fn held_activation_and_discard_race_selects_one_exact_disposition() -> Result { let fixture = Fixture::new(ObjectFormat::Sha256).await?; - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let mut published = 0; for n in 100..108 { let (prepared, root, budget) = empty(&fixture, [n; 16], "owner").await?; diff --git a/crates/canopy-server/src/packs/publication/tests/coordinator/preparation.rs b/crates/canopy-server/src/packs/publication/tests/coordinator/preparation.rs index 8f402a03..4a986910 100644 --- a/crates/canopy-server/src/packs/publication/tests/coordinator/preparation.rs +++ b/crates/canopy-server/src/packs/publication/tests/coordinator/preparation.rs @@ -1,10 +1,208 @@ use super::*; -async fn session(f: &Fixture, operation: [u8; 16]) -> Result> { - let started = f - .client() - .command::(&f.target, identity()?, f.begin(operation)) +#[tokio::test] +async fn cold_original_renewal_keeps_history_but_cannot_grant_previous_owner_custody() -> Result { + let mut observations = Vec::new(); + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let original_session = session(&f, [209; 16]).await?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let ticket = coordinator + .submit( + original_session + .ready_renew(identity()?, DEFAULT_LEASE_MS) + .await?, + ) + .await?; + let original = changed(ticket.wait().await)?.committed; + assert!(coordinator.close_and_drain().await.is_empty()); + let old_owner = original_session.lease.token.owner; + let (runtime, handle, client) = + super::super::durable_recovery::restore_owner(&f, &original_session.check).await?; + assert_ne!(handle.owner_fence(), old_owner); + // SQL still has the original operation and pin. Such historical facts + // are not an observation of the current admitted owner epoch. + let historical = client + .query::( + &f.target, + Some(original.receipt), + original_session.check.clone(), + ) + .await? + .output; + let restored = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let ticket = restored + .submit( + ReadyPreparation::restore(client, f.target.clone(), [209; 16], f.authority()) + .await?, + ) + .await?; + let outcome = changed(ticket.wait().await)?; + assert_eq!(outcome.committed, original); + observations.push((format, historical.is_some(), outcome.session.is_ok())); + assert!(restored.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + assert!( + observations.iter().all(|(_, _, usable)| !usable), + "previous owner became usable after cold restoration: {observations:?}" + ); + Ok(()) +} + +#[tokio::test] +async fn cold_takeover_fences_shared_old_session_and_restores_current_claim() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let old = session(&f, [210; 16]).await?; + let shared = old.clone(); + let (runtime, handle, client) = + super::super::durable_recovery::restore_owner(&f, &old.check).await?; + assert!(matches!( + old.check_owner().await, + Err(PreparationBaseError::Inactive) + )); + assert!(matches!( + shared.live_lease(), + Err(PreparationBaseError::Inactive) + )); + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let ready = ReadyPreparation::claim( + client.clone(), + f.target.clone(), + request_for(&old), + identity()?, + f.authority(), + ) .await?; + let result = changed(queue.submit(ready).await?.wait().await)?; + let current = result + .session + .as_ref() + .map_err(|e| format!("current claim: {e}"))?; + assert_eq!(current.lease.token.owner, handle.owner_fence()); + assert_ne!( + current.lease.token.artifact_operation, + old.lease.token.artifact_operation + ); + assert_eq!(current.lease.token.operation, old.lease.token.operation); + let restored = + ReadyPreparation::restore(client.clone(), f.target.clone(), [210; 16], f.authority()) + .await?; + let replay = changed(queue.submit(restored).await?.wait().await)?; + assert_eq!(replay.committed, result.committed); + replay + .session + .as_ref() + .map_err(|e| format!("restored current claim: {e}"))? + .live_lease()?; + // A fresh granted session can renew through the same registered service. + let renewed = changed( + queue + .submit(current.ready_renew(identity()?, DEFAULT_LEASE_MS).await?) + .await? + .wait() + .await, + )?; + assert_eq!( + renewed + .session + .as_ref() + .map_err(|e| format!("current renewal: {e}"))? + .lease + .token, + current.lease.token + ); + assert!(shared.live_lease().is_err()); + assert!(queue.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn missing_or_corrupt_owner_preserves_original_outcome_and_permanently_fences_session() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for corrupt in [false, true] { + let f = Fixture::new(format).await?; + let original_session = session(&f, [211; 16]).await?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let original = changed( + queue + .submit( + original_session + .ready_renew(identity()?, DEFAULT_LEASE_MS) + .await?, + ) + .await? + .wait() + .await, + )? + .committed; + let control_path = f.layout.control_path(f.target.cell_id().as_bytes()); + let (control, _) = f.layout.store().get_with_etag(&control_path).await?; + if corrupt { + f.layout + .store() + .put_overwrite(&control_path, bytes::Bytes::from_static(b"invalid control")) + .await?; + } else { + f.layout.store().delete(&control_path).await?; + } + let restored = + ReadyPreparation::restore(f.client(), f.target.clone(), [211; 16], f.authority()) + .await?; + let result = changed(queue.submit(restored).await?.wait().await)?; + assert_eq!(result.committed, original); + assert!(result.session.is_err()); + assert!(original_session.refresh(original.receipt).await.is_err()); + assert!( + original_session + .fenced + .load(std::sync::atomic::Ordering::Acquire) + ); + // Repairing the durable source does not undo a previously observed + // session fence. A fresh constructor must reacquire authority. + f.layout + .store() + .put_overwrite(&control_path, control) + .await?; + assert!(original_session.refresh(original.receipt).await.is_err()); + PreparationSession::open( + f.client(), + f.target.clone(), + original_session.check.clone(), + Some(original.receipt), + f.authority(), + ) + .await? + .live_lease()?; + assert!(queue.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +async fn session(f: &Fixture, operation: [u8; 16]) -> Result> { + let started = registered_preparation(f, operation).await?; let lease = lease(started.output)?; Ok(Arc::new( PreparationSession::open( @@ -12,6 +210,7 @@ async fn session(f: &Fixture, operation: [u8; 16]) -> Result Result { Ok(match kind { PreparationCommandKind::Claim => { - ReadyPreparation::claim(f.client(), f.target.clone(), request_for(s), mutation).await? + ReadyPreparation::claim( + f.client(), + f.target.clone(), + request_for(s), + mutation, + f.authority(), + ) + .await? } PreparationCommandKind::Renew => s.ready_renew(mutation, DEFAULT_LEASE_MS).await?, }) @@ -47,30 +253,34 @@ async fn replay( kind: PreparationCommandKind, mutation: MutationIdentity, ) -> Result> { - Ok(match kind { - PreparationCommandKind::Claim => { - f.client() - .command::(&f.target, mutation, request_for(s)) - .await? - } - PreparationCommandKind::Renew => { - f.client() - .command::(&f.target, mutation, request_for(s)) - .await? - } - }) + let saved = RegisteredCustody::load_latest(&f.client(), &f.target, s.lease.token.operation) + .await? + .ok_or("registered command missing")?; + assert_eq!(saved.evidence().identity(), mutation); + let result = saved.recover_preparation(&f.client()).await?; + let PreparationReply::Granted(lease) = &result.output else { + return Err("unexpected replay denial".into()); + }; + assert_eq!( + lease.token == s.lease.token, + kind == PreparationCommandKind::Renew + ); + Ok(result) } #[tokio::test] async fn bound_lease_commands_keep_exact_identity_and_original_floor_through_closed_uncertain_recovery() -> Result { for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { for kind in [PreparationCommandKind::Claim, PreparationCommandKind::Renew] { - for fault in [1, 2, 3] { + for fault in [1, 2, 3, 4, 5, 6] { let f = Fixture::new(format).await?; let s = session(&f, [196; 16]).await?; f.install_empty_root(1).await?; - let coordinator = - PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let mutation = identity()?; coordinator.fault_for_test(fault); let ticket = coordinator @@ -81,20 +291,31 @@ async fn bound_lease_commands_keep_exact_identity_and_original_floor_through_clo else { return Err("lease uncertainty".into()); }; - let PublicationError::Preparation(cellule_runtime::InvocationError::Pending( - evidence, - )) = &*error - else { - return Err("exact lease evidence".into()); + let (evidence, registrar) = match &*error { + PublicationError::Preparation(cellule_runtime::InvocationError::Pending( + evidence, + )) => ((**evidence).clone(), None), + PublicationError::Custody { evidence, source } => { + let CustodyError::Registration(source) = &**source else { + return Err("wrong registrar error".into()); + }; + let InvocationError::Pending(registrar) = &**source else { + return Err("registrar identity lost".into()); + }; + ((**evidence).clone(), Some((**registrar).clone())) + } + _ => return Err("exact lease evidence".into()), }; - let evidence = (**evidence).clone(); let original = match f.client().resolve(&evidence).await? { cellule_runtime::Resolution::Absent => None, cellule_runtime::Resolution::Committed(value) => Some(value.commit_sequence()), other => return Err(format!("unexpected {other:?}").into()), }; - assert_eq!(original.is_some(), fault != 1); - assert_eq!(coordinator.reservations_for_test().await, (1, 8192, 1)); + assert_eq!(original.is_some(), matches!(fault, 2 | 3)); + assert_eq!( + coordinator.reservations_for_test().await, + (1, super::super::super::custody::RESERVATION, 1) + ); drop(ticket); let retained = coordinator .pending([196; 16]) @@ -104,6 +325,12 @@ async fn bound_lease_commands_keep_exact_identity_and_original_floor_through_clo coordinator.recover(&retained).await?; let outcome = changed(timeout(Duration::from_secs(10), retained.wait()).await?)?; assert_eq!(outcome.kind, kind); + if let Some(registrar) = registrar { + assert!(matches!( + f.client().resolve(®istrar).await?, + cellule_runtime::Resolution::Committed(_) + )); + } assert_eq!(outcome.committed, replay(&f, &s, kind, mutation).await?); if let Some(sequence) = original { assert_eq!(outcome.committed.receipt.commit_sequence, sequence); @@ -151,6 +378,7 @@ async fn bound_lease_ready_admission_preserves_command_and_canceled_observer_ses uuid::Uuid::new_v4().into_bytes(), )?, PublicationLimits::default(), + f.publication_budget.clone(), )?; let rejected = foreign .submit(prepared) @@ -158,7 +386,11 @@ async fn bound_lease_ready_admission_preserves_command_and_canceled_observer_ses .err() .ok_or("foreign admitted")?; assert_eq!(rejected.reason, PublicationScheduleError::Foreign); - let coordinator = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let (release, entered) = coordinator.pause_for_test().await; let ticket = coordinator.submit(rejected.ready).await?; timeout(Duration::from_secs(5), entered).await??; @@ -191,7 +423,11 @@ async fn bound_lease_ready_admission_preserves_command_and_canceled_observer_ses .err() .ok_or("closed admitted")?; assert_eq!(refused.reason, PublicationScheduleError::Closed); - let other = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let other = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let retried = other.submit(refused.ready).await?; changed(timeout(Duration::from_secs(10), retried.wait()).await?)?; assert!(other.close_and_drain().await.is_empty()); @@ -206,8 +442,11 @@ async fn bound_lease_committed_recovery_keeps_receipt_when_fresh_custody_is_revo for mode in [0, 1, 2] { let f = Fixture::new(ObjectFormat::Sha256).await?; let s = session(&f, [198; 16]).await?; - let coordinator = - PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let mutation = identity()?; coordinator.fault_for_test(2); let ticket = coordinator @@ -245,8 +484,11 @@ async fn bound_lease_absent_recovery_rechecks_authority_and_claim_can_recover_ex for mode in [0, 1, 2] { let f = Fixture::new(ObjectFormat::Sha1).await?; let s = session(&f, [199; 16]).await?; - let coordinator = - PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; coordinator.fault_for_test(1); let ticket = coordinator .submit(ready(&f, &s, kind, identity()?).await?) @@ -309,18 +551,34 @@ async fn bound_lease_ready_rejects_invalid_context_size_duration_and_never_reviv let mut wrong = request_for(&s); wrong.check.token.repository = uuid::Uuid::new_v4().into_bytes(); assert!( - ReadyPreparation::claim(f.client(), f.target.clone(), wrong, identity()?) - .await - .is_err() + ReadyPreparation::claim( + f.client(), + f.target.clone(), + wrong, + identity()?, + f.authority(), + ) + .await + .is_err() ); let mut huge = request_for(&s); huge.check.actor = "x".repeat(8192); assert!( - ReadyPreparation::claim(f.client(), f.target.clone(), huge, identity()?) - .await - .is_err() + ReadyPreparation::claim( + f.client(), + f.target.clone(), + huge, + identity()?, + f.authority(), + ) + .await + .is_err() ); - let coordinator = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; coordinator.fault_for_test(2); let ticket = coordinator .submit(s.ready_renew(identity()?, DEFAULT_LEASE_MS).await?) @@ -371,8 +629,11 @@ async fn bound_lease_claim_after_actual_owner_restore_uses_new_fence_and_preserv ) .await?; let client = CellClient::local(Arc::clone(&f.registry), handle.clone()); - let coordinator = - PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let mutation = identity()?; coordinator.fault_for_test(2); let ticket = coordinator @@ -382,6 +643,7 @@ async fn bound_lease_claim_after_actual_owner_restore_uses_new_fence_and_preserv f.target.clone(), request_for(&original), mutation, + f.authority(), ) .await?, ) @@ -397,9 +659,11 @@ async fn bound_lease_claim_after_actual_owner_restore_uses_new_fence_and_preserv .ok_or("restored Claim retained")?; coordinator.recover(&retained).await?; let outcome = changed(timeout(Duration::from_secs(10), retained.wait()).await?)?; - let replay = client - .command::(&f.target, mutation, request_for(&original)) - .await?; + let original_command = RegisteredCustody::load_latest(&client, &f.target, old.operation) + .await? + .ok_or("claim journal missing")?; + assert_eq!(original_command.evidence().identity(), mutation); + let replay = original_command.recover_preparation(&client).await?; assert_eq!(outcome.committed, replay); let current = outcome.session.map_err(|e| e.to_string())?; let next = current.live_lease()?.0; diff --git a/crates/canopy-server/src/packs/publication/tests/custody.rs b/crates/canopy-server/src/packs/publication/tests/custody.rs new file mode 100644 index 00000000..193fe140 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/custody.rs @@ -0,0 +1,771 @@ +//! Real Cell receipts, pre-admission recovery and the original command's atomic result. +use super::{publishing::edit, *}; +use cellule_runtime::{Committed, Resolution}; +use tokio::time::Duration; + +async fn prepare(f: &Fixture, action: CustodyAction) -> Result { + Ok(PreparedCustody::prepare(&f.client(), &f.target, action, identity()?).await?) +} +async fn execute(f: &Fixture, action: CustodyAction) -> Result> { + Ok(prepare(f, action) + .await? + .register(&f.client(), identity()?) + .await? + .recover(&f.client()) + .await?) +} +fn token(output: &CustodyReply) -> Result { + match output { + CustodyReply::Preparation(PreparationReply::Granted(lease)) => Ok(lease.token), + CustodyReply::Staging(StagingReply::Granted(lease)) => Ok(lease.token), + other => Err(format!("expected custody grant: {other:?}").into()), + } +} +async fn expire(identity: MutationIdentity) -> Result { + let now = sql::now(0)?; + if now <= identity.expires_at_ms { + tokio::time::sleep(Duration::from_millis(u64::try_from( + identity.expires_at_ms - now + 1, + )?)) + .await; + } + Ok(()) +} + +#[tokio::test] +async fn owned_registrar_loss_preserves_both_originals_and_recovers_without_new_namespace() -> Result +{ + use crate::packs::publication::custody::OwnedCustody; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [4, 5, 6] { + let f = Fixture::new(format).await?; + let operation = [fault + 180; 16]; + let owned = OwnedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginPreparation(f.begin(operation)), + identity()?, + ) + .await?; + let original = owned.evidence().clone(); + let registrar = owned + .registration_evidence() + .ok_or("registrar missing")? + .clone(); + assert_ne!(original, registrar); + let dispatch = owned.clone(); + let client = f.client(); + let result = + tokio::spawn( + async move { dispatch.invoke(&client, false, fault, || Ok(())).await }, + ) + .await; + if fault == 6 { + assert!(result.is_err_and(|error| error.is_panic())); + } else { + let Err(CustodyError::Registration(error)) = result? else { + return Err("registrar fault did not preserve uncertainty".into()); + }; + let InvocationError::Pending(evidence) = *error else { + return Err("registrar fault returned terminal evidence".into()); + }; + assert_eq!(*evidence, registrar); + } + assert_eq!(owned.evidence(), &original); + assert_eq!(owned.registration_evidence(), Some(®istrar)); + assert_eq!(f.counts().await?, (0, 0)); + assert!(matches!( + f.client().resolve(&original).await?, + Resolution::Absent + )); + let registered = + RegisteredCustody::load_latest(&f.client(), &f.target, operation).await?; + assert_eq!(registered.is_some(), fault != 4); + if let Some(registered) = registered { + assert_eq!(registered.evidence(), &original); + } + let committed = owned.invoke(&f.client(), true, 0, || Ok(())).await??; + assert_eq!( + token(&committed.output)?.artifact_operation, + artifact_number(1) + ); + assert_eq!(f.counts().await?, (1, 1)); + assert!(matches!( + f.client().resolve(®istrar).await?, + Resolution::Committed(_) + )); + drop(owned); + let restored = OwnedCustody::restore(&f.client(), &f.target, operation).await?; + assert_eq!(restored.evidence(), &original); + assert!(restored.registration_evidence().is_none()); + // Recorded knowledge precedes a fresh execution guard; recovery + // must not execute another Begin or allocate another namespace. + let replay = restored + .invoke(&f.client(), true, 0, || { + Err(Error::Command("fresh custody denied")) + }) + .await??; + assert_eq!(replay, committed); + assert_eq!(f.counts().await?, (1, 1)); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn intent_precedes_namespace_and_refuses_unregistered_or_losing_execution() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let operation = [231; 16]; + let action = CustodyAction::BeginPreparation(f.begin(operation)); + let original = prepare(&f, action.clone()).await?; + let command = original.command_for_test(&f.client())?; + assert!(matches!( + command.clone().execute().await, + Err(InvocationError::NotStarted(_)) + )); + assert!(matches!( + f.client().resolve(command.evidence()).await?, + Resolution::Absent + )); + let loser = prepare(&f, action.clone()).await?; + let registered = original.register(&f.client(), identity()?).await?; + assert_eq!(f.counts().await?, (0, 0)); + assert!(!registered.settled()); + assert!( + matches!(prepare(&f, action).await, Err(error) if error.downcast_ref::().is_some_and(|e| matches!(e, CustodyError::Unsettled(_)))) + ); + assert!(loser.register(&f.client(), identity()?).await.is_err()); + let losing = loser.command_for_test(&f.client())?; + assert!(matches!( + losing.clone().execute().await, + Err(InvocationError::NotStarted(_)) + )); + assert!(matches!( + f.client().resolve(losing.evidence()).await?, + Resolution::Absent + )); + // Discard all capabilities: discovery preserves the first SDK identity. + drop(registered); + drop(original); + let discovered = RegisteredCustody::load_latest(&f.client(), &f.target, operation) + .await? + .ok_or("intent missing")?; + assert_eq!(discovered.evidence(), command.evidence()); + let accepted = discovered.recover(&f.client()).await?; + let granted = token(&accepted.output)?; + assert_eq!(granted.attempt, accepted.receipt.commit_sequence); + assert_eq!(granted.owner, f.handle.owner_fence()); + assert_eq!(granted.artifact_operation, artifact_number(1)); + assert_eq!(f.counts().await?, (1, 1)); + assert_eq!(discovered.recover(&f.client()).await?, accepted); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn registration_and_late_phase_failure_keep_sdk_absent_and_exact_retry() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let original = prepare(&f, CustodyAction::BeginStaging(f.begin([232; 16]))).await?; + let registration = f + .client() + .prepare_command::( + &f.target, + identity()?, + original.intent_for_test(), + ) + .await?; + edit(&f, "CREATE TRIGGER custody_registration_fault BEFORE INSERT ON catalog_custody_commands BEGIN SELECT RAISE(ABORT,'late custody registration failure'); END").await?; + assert!( + matches!(registration.clone().execute().await, Err(InvocationError::NotStarted(error)) if format!("{error:?}").contains("late custody registration failure")) + ); + assert!(matches!( + f.client().resolve(registration.evidence()).await?, + Resolution::Absent + )); + assert!( + RegisteredCustody::load_latest(&f.client(), &f.target, [232; 16]) + .await? + .is_none() + ); + assert_eq!(f.counts().await?, (0, 0)); + edit(&f, "DROP TRIGGER custody_registration_fault").await?; + registration.execute().await?; + // Losing the registration acknowledgement is recovered by its durable + // pointer, without preparing another original custody command. + let registered = RegisteredCustody::load_latest(&f.client(), &f.target, [232; 16]) + .await? + .ok_or("registration absent")?; + assert_eq!(registered.evidence(), original.evidence()); + edit(&f, "CREATE TRIGGER custody_phase_fault BEFORE UPDATE OF phase ON catalog_custody_commands BEGIN SELECT RAISE(ABORT,'late custody phase failure'); END").await?; + let command = original.command_for_test(&f.client())?; + assert!( + matches!(command.clone().execute().await, Err(InvocationError::NotStarted(error)) if format!("{error:?}").contains("late custody phase failure")) + ); + assert!(matches!( + f.client().resolve(command.evidence()).await?, + Resolution::Absent + )); + assert_eq!(f.counts().await?, (0, 0)); + assert!( + StagingAdmission::load(&f.client(), &f.target, [232; 16]) + .await? + .is_none() + ); + edit(&f, "DROP TRIGGER custody_phase_fault").await?; + let committed = registered.recover(&f.client()).await?; + assert_eq!( + token(&committed.output)?.artifact_operation, + artifact_number(1) + ); + assert_eq!(f.counts().await?, (1, 1)); + for sql in [ + "UPDATE catalog_custody_commands SET intent=x'01'", + "UPDATE catalog_custody_commands SET phase=NULL", + "DELETE FROM catalog_custody_commands", + "INSERT OR REPLACE INTO catalog_custody_commands SELECT * FROM catalog_custody_commands", + ] { + assert!(edit(&f, sql).await.is_err()); + } + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn every_custody_transition_retains_its_original_result_across_successors() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let operation = [233; 16]; + let mut known = Vec::new(); + let mut next = CustodyAction::BeginStaging(f.begin(operation)); + for step in 0..7 { + let original = prepare(&f, next).await?; + let registered = original.register(&f.client(), identity()?).await?; + let committed = registered.recover(&f.client()).await?; + let current = token(&committed.output)?; + known.push((registered, committed)); + next = match step { + 0 => CustodyAction::RenewStaging(request(current)), + 1 => CustodyAction::ClaimStaging(request(current)), + 2 => CustodyAction::BindStaging(check(current)), + 3 => CustodyAction::RenewPreparation(request(current)), + 4 => CustodyAction::ClaimPreparation(request(current)), + _ => CustodyAction::BeginPreparation(f.begin(operation)), + }; + } + for (registered, committed) in known { + assert_eq!(registered.recover(&f.client()).await?, committed); + } + let operation_count = f + .handle + .query(0, 1024, |db| { + Ok( + db.query_row("SELECT count(*) FROM catalog_custody_commands", [], |r| { + r.get::<_, u64>(0).map(|value| value.to_be_bytes().to_vec()) + })?, + ) + }) + .await?; + assert_eq!(u64::from_be_bytes(operation_count.try_into().unwrap()), 7); + assert_eq!(f.counts().await?, (1, 3)); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn denied_begin_is_original_knowledge_after_sdk_expiry_and_authority_changes() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let operation = [234; 16]; + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let original = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginPreparation(f.begin(operation)), + mutation, + ) + .await?; + let registered = original.register(&f.client(), identity()?).await?; + edit(&f, "UPDATE repository_identity SET owner='other'").await?; + let committed = match registered.recover(&f.client()).await { + Err(InvocationError::Rejected(committed)) => *committed, + other => return Err(format!("expected original unauthorized denial: {other:?}").into()), + }; + assert_eq!( + committed.output, + CustodyReply::Preparation(PreparationReply::Denied(PreparationDenial::Unauthorized)) + ); + assert_eq!(f.counts().await?, (0, 0)); + expire(mutation).await?; + assert!(matches!( + f.client().resolve(original.evidence()).await?, + Resolution::Expired + )); + edit(&f, "UPDATE repository_identity SET owner='owner'").await?; + let next = execute(&f, CustodyAction::BeginPreparation(f.begin(operation))).await?; + assert_eq!(token(&next.output)?.artifact_operation, artifact_number(1)); + assert!( + matches!(registered.recover(&f.client()).await, Err(InvocationError::Rejected(value)) if *value == committed) + ); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn cold_owner_restoration_recovers_claim_and_renew_receipts_without_reviving_custody() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for claim in [false, true] { + let f = Fixture::new(format).await?; + let operation = [235; 16]; + let started = execute(&f, CustodyAction::BeginPreparation(f.begin(operation))).await?; + let old = token(&started.output)?; + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let action = if claim { + CustodyAction::ClaimPreparation(request(old)) + } else { + CustodyAction::RenewPreparation(request(old)) + }; + let original = + PreparedCustody::prepare(&f.client(), &f.target, action, mutation).await?; + let registered = original.register(&f.client(), identity()?).await?; + let accepted = registered.recover(&f.client()).await?; + let token = token(&accepted.output)?; + let (runtime, handle, client) = + super::durable_recovery::restore_owner(&f, &check(token)).await?; + expire(mutation).await?; + assert!(matches!( + client.resolve(original.evidence()).await?, + Resolution::Expired + )); + edit_restored(&handle, "UPDATE repository_identity SET owner='other'").await?; + let recovered = RegisteredCustody::load_latest(&client, &f.target, operation) + .await? + .ok_or("cold custody receipt absent")?; + assert_eq!(recovered.evidence(), original.evidence()); + assert_eq!(recovered.recover(&client).await?, accepted); + assert_ne!(token.owner, handle.owner_fence()); + assert!( + client + .query::(&f.target, Some(accepted.receipt), check(token)) + .await? + .output + .is_none() + ); + runtime.shutdown().await?; + } + } + Ok(()) +} +async fn edit_restored(handle: &CellHandle, sql: &'static str) -> Result { + super::publishing::edit_handle(handle, sql).await +} + +#[tokio::test] +async fn expired_unsettled_identity_cannot_be_replaced_or_reported_as_absent() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let operation = [236; 16]; + let action = CustodyAction::BeginStaging(f.begin(operation)); + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let original = + PreparedCustody::prepare(&f.client(), &f.target, action.clone(), mutation).await?; + let registered = original.register(&f.client(), identity()?).await?; + expire(mutation).await?; + assert!( + matches!(registered.recover(&f.client()).await, Err(InvocationError::Pending(evidence)) if *evidence == *original.evidence()) + ); + assert!( + matches!(PreparedCustody::prepare(&f.client(), &f.target, action, identity()?).await, Err(CustodyError::Unsettled(evidence)) if *evidence == *original.evidence()) + ); + assert_eq!(f.counts().await?, (0, 0)); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn denied_renewal_preserves_knowledge_and_exact_successor_claim_survives_reaping() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for staging in [false, true] { + let f = Fixture::new(format).await?; + let input = f.begin([237; 16]); + let begin = if staging { + CustodyAction::BeginStaging(input) + } else { + CustodyAction::BeginPreparation(input) + }; + let first = execute(&f, begin).await?; + let original = token(&first.output)?; + let claim = if staging { + CustodyAction::ClaimStaging(request(original)) + } else { + CustodyAction::ClaimPreparation(request(original)) + }; + let second = execute(&f, claim).await?; + let prior = token(&second.output)?; + assert_ne!(prior, original); + edit(&f, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0").await?; + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let action = if staging { + CustodyAction::RenewStaging(request(prior)) + } else { + CustodyAction::RenewPreparation(request(prior)) + }; + let denied = PreparedCustody::prepare(&f.client(), &f.target, action, mutation) + .await? + .register(&f.client(), identity()?) + .await?; + let original_denial = match denied.recover(&f.client()).await { + Err(InvocationError::Rejected(value)) => *value, + other => return Err(format!("expected original expired denial: {other:?}").into()), + }; + let expected = if staging { + CustodyReply::Staging(StagingReply::Denied(PreparationDenial::Expired)) + } else { + CustodyReply::Preparation(PreparationReply::Denied(PreparationDenial::Expired)) + }; + assert_eq!(original_denial.output, expected); + f.client() + .command::( + &f.target, + identity()?, + MaintenanceRequest { + repository: f.repository, + actor: "owner".into(), + owner: f.handle.owner_fence(), + }, + ) + .await?; + assert_eq!(f.counts().await?, (0, 0)); + // A forged token cannot borrow the authenticated previous grant. + let mut forged = prior; + forged.request_digest[0] ^= 1; + assert!( + PreparedCustody::prepare( + &f.client(), + &f.target, + if staging { + CustodyAction::ClaimStaging(request(forged)) + } else { + CustodyAction::ClaimPreparation(request(forged)) + }, + identity()? + ) + .await + .is_err() + ); + let claimed = execute( + &f, + if staging { + CustodyAction::ClaimStaging(request(prior)) + } else { + CustodyAction::ClaimPreparation(request(prior)) + }, + ) + .await?; + let next = token(&claimed.output)?; + assert_ne!(next, prior); + assert_eq!(next.owner, f.handle.owner_fence()); + assert_eq!(next.attempt, claimed.receipt.commit_sequence); + assert_eq!(next.artifact_operation, artifact_number(3)); + assert_eq!(f.counts().await?, (1, 1)); + expire(mutation).await?; + assert!(matches!( + f.client().resolve(denied.evidence()).await?, + Resolution::Expired + )); + assert!( + matches!(denied.recover(&f.client()).await, Err(InvocationError::Rejected(value)) if *value == original_denial) + ); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn corrupted_metadata_blocks_sdk_fallback_and_journal_queries_are_indexed() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let operation = [238; 16]; + let registered = prepare(&f, CustodyAction::BeginPreparation(f.begin(operation))) + .await? + .register(&f.client(), identity()?) + .await?; + registered.recover(&f.client()).await?; + let plans = f.handle.query(0, 4096, |db| { + let mut all = String::new(); + for sql in [ + "EXPLAIN QUERY PLAN SELECT intent,phase FROM catalog_custody_commands WHERE purpose=0 AND operation=zeroblob(16) ORDER BY step DESC LIMIT 1", + "EXPLAIN QUERY PLAN SELECT operation FROM catalog_custody_commands WHERE phase IS NULL AND stopped IS NULL LIMIT 1024", + "EXPLAIN QUERY PLAN SELECT intent,phase FROM catalog_custody_commands INDEXED BY catalog_custody_grants WHERE purpose=0 AND operation=zeroblob(16) AND granted_incarnation=zeroblob(16) AND granted_attempt=1 ORDER BY step DESC LIMIT 1", + ] { + let mut statement = db.prepare(sql)?; + let mut rows = statement.query([])?; + while let Some(row) = rows.next()? { all.push_str(&row.get::<_,String>(3)?); all.push('\n'); } + } + Ok(all.into_bytes()) + }).await?; + let plans = String::from_utf8(plans)?; + assert!(plans.contains("PRIMARY KEY"), "{plans}"); + assert!(plans.contains("catalog_custody_pending"), "{plans}"); + assert!(plans.contains("catalog_custody_grants"), "{plans}"); + assert!(matches!( + f.client().resolve(registered.evidence()).await?, + Resolution::Committed(_) + )); + edit(&f, "DROP TRIGGER catalog_custody_identity_immutable").await?; + f.handle + .execute( + identity()?, + Digest::from_bytes([239; 32]), + sql::now(0)?, + 4096, + 0, + move |tx| { + let mut bytes: Vec = tx.query_row( + "SELECT intent FROM catalog_custody_commands WHERE operation=?1", + [operation.as_slice()], + |row| row.get(0), + )?; + *bytes + .last_mut() + .ok_or(cellule_runtime::Error::Command("custody body absent"))? ^= 1; + tx.execute( + "UPDATE catalog_custody_commands SET intent=?1 WHERE operation=?2", + rusqlite::params![bytes, operation.as_slice()], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert!( + RegisteredCustody::load_latest(&f.client(), &f.target, operation) + .await + .is_err() + ); + assert!( + matches!(registered.recover(&f.client()).await, Err(InvocationError::Pending(evidence)) if *evidence == *registered.evidence()) + ); + assert_eq!(f.counts().await?, (1, 1)); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn initialized_generation_grants_fit_the_journal_and_preserve_joint_roots() -> Result { + use canopy_object_storage::artifact::ArtifactStore; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let (prepared, _root, _budget) = super::initialization::empty(&f, [240; 16], store).await?; + let proof = prepared.empty_ref_initialization().await?; + let (initialization, _) = + super::initialization::registered(&f, &prepared, proof, identity()?).await?; + let initialized = initialization.execute().await?; + let InitializationReply::Initialized(fact) = initialized.output else { + return Err("initialization failed".into()); + }; + assert!(fact.catalog.is_some() && fact.refs.is_some()); + let original = prepare(&f, CustodyAction::BeginPreparation(f.begin([241; 16]))) + .await? + .register(&f.client(), identity()?) + .await?; + let committed = original.recover_preparation(&f.client()).await?; + let PreparationReply::Granted(lease) = committed.output else { + return Err("grant absent".into()); + }; + assert_eq!(lease.base, *fact); + let renewed = execute(&f, CustodyAction::RenewPreparation(request(lease.token))).await?; + let CustodyReply::Preparation(PreparationReply::Granted(renewed)) = renewed.output else { + return Err("renewal absent".into()); + }; + assert_eq!(renewed.base, *fact); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn ignored_registration_or_phase_write_cannot_commit_without_domain_knowledge() -> Result { + for phase in [false, true] { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let operation = [242; 16]; + let original = prepare(&f, CustodyAction::BeginPreparation(f.begin(operation))).await?; + if phase { + original.register(&f.client(), identity()?).await?; + } + let event = if phase { "UPDATE OF phase" } else { "INSERT" }; + edit(&f, &format!("CREATE TRIGGER custody_ignore_fault BEFORE {event} ON catalog_custody_commands BEGIN SELECT RAISE(IGNORE); END")).await?; + if phase { + let command = original.command_for_test(&f.client())?; + assert!( + matches!(command.clone().execute().await, Err(InvocationError::NotStarted(error)) if format!("{error:?}").contains("publication changed unexpected rows")) + ); + assert!(matches!( + f.client().resolve(command.evidence()).await?, + Resolution::Absent + )); + } else { + let command = f + .client() + .prepare_command::( + &f.target, + identity()?, + original.intent_for_test(), + ) + .await?; + assert!( + matches!(command.clone().execute().await, Err(InvocationError::NotStarted(error)) if format!("{error:?}").contains("publication changed unexpected rows")) + ); + assert!(matches!( + f.client().resolve(command.evidence()).await?, + Resolution::Absent + )); + assert!( + RegisteredCustody::load_latest(&f.client(), &f.target, operation) + .await? + .is_none() + ); + } + assert_eq!(f.counts().await?, (0, 0)); + edit(&f, "DROP TRIGGER custody_ignore_fault").await?; + original + .register(&f.client(), identity()?) + .await? + .recover(&f.client()) + .await?; + assert_eq!(f.counts().await?, (1, 1)); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn registered_unexecuted_begin_survives_deleted_sqlite_and_cold_owner_restore() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for staging in [false, true] { + let f = Fixture::new(format).await?; + let operation = [243; 16]; + let action = if staging { + CustodyAction::BeginStaging(f.begin(operation)) + } else { + CustodyAction::BeginPreparation(f.begin(operation)) + }; + let prepared = prepare(&f, action.clone()).await?; + let original = prepared.evidence().clone(); + prepared.register(&f.client(), identity()?).await?; + assert_eq!(f.counts().await?, (0, 0)); + assert!(matches!( + f.client().resolve(&original).await?, + Resolution::Absent + )); + drop(prepared); + let (runtime, handle, client) = + super::durable_recovery::restore_owner_fence(&f, f.handle.owner_fence()).await?; + assert_eq!(super::counts(&handle).await?, (0, 0)); + let recovered = RegisteredCustody::load_latest(&client, &f.target, operation) + .await? + .ok_or("pre-namespace intent missing")?; + assert_eq!(recovered.evidence(), &original); + assert_eq!(recovered.action()?, action); + assert!(matches!( + client.resolve(&original).await?, + Resolution::Absent + )); + let committed = recovered.recover(&client).await?; + let admitted = token(&committed.output)?; + assert_eq!(admitted.owner, handle.owner_fence()); + assert_eq!(admitted.attempt, committed.receipt.commit_sequence); + assert_eq!(admitted.artifact_operation, artifact_number(1)); + assert_eq!(super::counts(&handle).await?, (1, 1)); + assert_eq!(recovered.recover(&client).await?, committed); + runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn denied_claim_keeps_its_original_receipt_after_sdk_expiry_and_cold_restore() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for staging in [false, true] { + let f = Fixture::new(format).await?; + let operation = [244; 16]; + let begun = execute( + &f, + if staging { + CustodyAction::BeginStaging(f.begin(operation)) + } else { + CustodyAction::BeginPreparation(f.begin(operation)) + }, + ) + .await?; + let old = token(&begun.output)?; + let accepted = execute( + &f, + if staging { + CustodyAction::ClaimStaging(request(old)) + } else { + CustodyAction::ClaimPreparation(request(old)) + }, + ) + .await?; + let successor = token(&accepted.output)?; + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let action = if staging { + CustodyAction::ClaimStaging(request(old)) + } else { + CustodyAction::ClaimPreparation(request(old)) + }; + let denied = PreparedCustody::prepare(&f.client(), &f.target, action, mutation) + .await? + .register(&f.client(), identity()?) + .await?; + let original = match denied.recover(&f.client()).await { + Err(InvocationError::Rejected(value)) => *value, + other => return Err(format!("expected original stale Claim: {other:?}").into()), + }; + let expected = if staging { + CustodyReply::Staging(StagingReply::Denied(PreparationDenial::Stale)) + } else { + CustodyReply::Preparation(PreparationReply::Denied(PreparationDenial::Stale)) + }; + assert_eq!(original.output, expected); + assert_eq!(f.counts().await?, (1, 2)); + let evidence = denied.evidence().clone(); + let (runtime, handle, client) = + super::durable_recovery::restore_owner_fence(&f, successor.owner).await?; + expire(mutation).await?; + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Expired + )); + let restored = RegisteredCustody::load_latest(&client, &f.target, operation) + .await? + .ok_or("cold Claim denial missing")?; + assert_eq!(restored.evidence(), &evidence); + assert!( + matches!(restored.recover(&client).await, Err(InvocationError::Rejected(value)) if *value == original) + ); + assert_eq!(super::counts(&handle).await?, (1, 2)); + runtime.shutdown().await?; + } + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/custody_stop.rs b/crates/canopy-server/src/packs/publication/tests/custody_stop.rs new file mode 100644 index 00000000..8ce17155 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/custody_stop.rs @@ -0,0 +1,781 @@ +//! Separate logical retirement, actual receiver races and bounded maintenance. +use super::{publishing::edit, staging_service::restore::head_expiring, *}; +use cellule_runtime::Resolution; +use tokio::time::{Duration, timeout}; + +async fn expired(value: &cellule_runtime::PendingMutation) -> Result { + let now = sql::now(0)?; + if now <= value.identity().expires_at_ms { + tokio::time::sleep(Duration::from_millis( + (value.identity().expires_at_ms - now + 1) as u64, + )) + .await; + } + Ok(()) +} +async fn registered(f: &Fixture, kind: u8) -> Result { + Ok( + RegisteredCustody::load_latest(&f.client(), &f.target, [230 + kind; 16]) + .await? + .ok_or("original missing")?, + ) +} +fn stopped(state: PublicationState) -> Result { + match state { + PublicationState::Finished(Ok(PublicationOutcome::CustodyStop(value))) => Ok(*value), + other => Err(format!("stop outcome: {other:?}").into()), + } +} +async fn observed(ticket: &PublicationTicket) -> Result { + Ok(timeout(Duration::from_secs(10), ticket.wait()).await?) +} + +#[tokio::test] +async fn stop_all_seven_expired_originals_preserves_unknown_outcomes_and_allows_explicit_successors() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in 0..7 { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, kind, false, true).await?; + expired(&evidence).await?; + let original = registered(&f, kind).await?; + let counts = f.counts().await?; + let ready = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + let stop_evidence = ready.evidence().clone(); + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let ticket = queue.submit(ready).await?; + let outcome = stopped(observed(&ticket).await?)?; + assert_eq!(outcome.original, evidence); + assert_eq!(outcome.invocation, stop_evidence); + assert_eq!( + outcome + .committed + .as_ref() + .ok_or("stop invocation lost")? + .output, + CustodyStopReply::Stopped + ); + let fact = outcome.stop.ok_or("stop fact missing")?; + assert_eq!( + fact.receipt, + outcome.committed.ok_or("stop receipt missing")?.receipt + ); + assert_eq!(fact.owner, f.handle.owner_fence()); + assert_eq!(f.counts().await?, counts); + let loaded = registered(&f, kind).await?; + assert_eq!(loaded.evidence(), &evidence); + assert!(loaded.closed()); + assert!(!loaded.settled()); + assert_eq!(loaded.stop_fact(), Some(fact.clone())); + assert!( + matches!(loaded.recover(&f.client()).await, Err(InvocationError::Pending(value)) if *value == evidence) + ); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + assert!( + matches!(loaded.ready_stop(f.client(), identity()?, &f.authority()).await, Err(CustodyError::Stopped(value)) if *value == fact) + ); + let staging = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let cold = staging + .submit( + ReadyStaging::restore(f.client(), f.target.clone(), [230 + kind; 16]).await?, + ) + .map_err(|(error, _)| error)?; + let StagingState::Fenced(error) = + timeout(Duration::from_secs(10), cold.wait_terminal()).await? + else { + return Err("stopped cold stage not fenced".into()); + }; + assert!( + matches!(&*error, StagingError::Custody { source, .. } if matches!(&**source, CustodyError::Stopped(_))) + ); + assert_eq!(cold.restored_evidence(), Some(&evidence)); + assert!(cold.restored_outcome().is_none()); + assert_eq!(staging.stats().command_bytes, 0); + assert!(staging.close_and_drain().await.is_empty()); + // A deliberate successor keeps logical actor/digest continuity and + // does not erase or assign an invented result to the stopped original. + let next = + PreparedCustody::prepare(&f.client(), &f.target, loaded.action()?, identity()?) + .await?; + assert_ne!(next.evidence(), &evidence); + let next = next.register(&f.client(), identity()?).await?; + assert!(!next.closed()); + assert_eq!(original.stop_fact(), None); // Old DTO is not fresh state. + assert!(queue.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn stop_receiver_refuses_live_forged_and_stale_owner_proofs_and_preserves_accepted_originals() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 0, false, false).await?; + let original = registered(&f, 0).await?; + let ready = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + assert!(matches!(ready.command_for_test().execute().await, + Err(InvocationError::Rejected(value)) if value.output == CustodyStopReply::Denied(PreparationDenial::Conflict))); + assert!(!registered(&f, 0).await?.closed()); + let forged = f + .client() + .prepare_command::( + &f.target, + identity()?, + ready.input_for_test()?.tamper_for_test(), + ) + .await?; + assert!(matches!(forged.execute().await, + Err(InvocationError::Rejected(value)) if value.output == CustodyStopReply::Denied(PreparationDenial::Unauthorized))); + let ready = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + let expected = original.recover(&f.client()).await?; + // Execute won before stop. Even revoked access must not rewrite history. + edit(&f, "UPDATE repository_identity SET owner='other'").await?; + let command = ready.command_for_test(); + assert_eq!(command.execute().await?.output, CustodyStopReply::Settled); + assert_eq!( + registered(&f, 0).await?.recover(&f.client()).await?, + expected + ); + assert!(registered(&f, 0).await?.stop_fact().is_none()); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Committed(_) + )); + f.runtime.shutdown().await?; + + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 0, false, true).await?; + let original = registered(&f, 0).await?; + let ready = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + let input = ready.input_for_test()?; + let (runtime, handle, client) = + super::durable_recovery::restore_owner_fence(&f, f.handle.owner_fence()).await?; + expired(&evidence).await?; + let stale = client + .prepare_command::(&f.target, identity()?, input) + .await?; + assert!(matches!(stale.execute().await, + Err(InvocationError::Rejected(value)) if value.output == CustodyStopReply::Denied(PreparationDenial::Stale))); + let original = RegisteredCustody::load_latest(&client, &f.target, [230; 16]) + .await? + .ok_or("restored head")?; + assert!(!original.closed()); + let fresh = original + .ready_stop(client.clone(), identity()?, &f.authority()) + .await?; + assert_eq!( + fresh.command_for_test().execute().await?.output, + CustodyStopReply::Stopped + ); + assert_eq!( + RegisteredCustody::load_latest(&client, &f.target, [230; 16]) + .await? + .ok_or("stopped head")? + .stop_fact() + .ok_or("actual stop")? + .owner, + handle.owner_fence() + ); + runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn stop_first_writer_fact_is_immutable_and_is_not_an_unexecuted_invocations_receipt() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 4, false, true).await?; + expired(&evidence).await?; + let original = registered(&f, 4).await?; + let first = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + let second = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + let first_receipt = first.command_for_test().execute().await?.receipt; + let invocation = second.evidence().clone(); + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let outcome = stopped(observed(&queue.submit(second).await?).await?)?; + assert_eq!(outcome.invocation, invocation); + assert!(outcome.committed.is_none()); + assert_eq!( + outcome.stop.ok_or("first writer fact")?.receipt, + first_receipt + ); + assert!(matches!( + f.client().resolve(&invocation).await?, + Resolution::Absent + )); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + for sql in [ + "UPDATE catalog_custody_commands SET stopped=NULL", + "UPDATE catalog_custody_commands SET stopped=x'01'", + "UPDATE catalog_custody_commands SET phase=x'01'", + "DELETE FROM catalog_custody_commands", + ] { + assert!(edit(&f, sql).await.is_err(), "{sql}"); + } + assert!(queue.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn stop_late_failure_and_ignored_write_rollback_marker_and_sdk_acceptance_before_exact_retry() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for ignored in [false, true] { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 0, false, true).await?; + expired(&evidence).await?; + let original = registered(&f, 0).await?; + let ready = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + let command = ready.command_for_test(); + let action = if ignored { + "RAISE(IGNORE)" + } else { + "RAISE(ABORT,'late stop fault')" + }; + edit(&f, &format!("CREATE TRIGGER stop_fault BEFORE UPDATE OF stopped ON catalog_custody_commands BEGIN SELECT {action}; END")).await?; + assert!(matches!( + command.clone().execute().await, + Err(InvocationError::NotStarted(_)) + )); + assert!(matches!( + f.client().resolve(command.evidence()).await?, + Resolution::Absent + )); + assert!(!registered(&f, 0).await?.closed()); + edit(&f, "DROP TRIGGER stop_fault").await?; + let result = command.clone().execute().await?; + assert_eq!(result.output, CustodyStopReply::Stopped); + assert_eq!( + registered(&f, 0) + .await? + .stop_fact() + .ok_or("stop missing")? + .receipt, + result.receipt + ); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn stop_service_reply_loss_and_panic_keep_bounded_originals_through_closed_recovery() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in 0..=3 { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 0, false, true).await?; + expired(&evidence).await?; + let mut invocation_identity = identity()?; + if fault > 1 { + invocation_identity.expires_at_ms = invocation_identity.issued_at_ms + 1_000; + } + let ready = registered(&f, 0) + .await? + .ready_stop(f.client(), invocation_identity, &f.authority()) + .await?; + let invocation = ready.evidence().clone(); + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + if fault == 0 { + edit( + &f, + "ALTER TABLE catalog_custody_commands RENAME TO custody_stop_query_fault", + ) + .await?; + } else { + queue.fault_for_test(fault); + } + let ticket = queue.submit(ready).await?; + assert!(matches!( + observed(&ticket).await?, + PublicationState::Uncertain(_) + )); + let stats = queue.stats().await; + assert_eq!(stats.maintenance, 1); + assert_eq!(stats.foreground, 0); + assert_eq!(stats.command_bytes, 8 << 10); + drop(ticket); + let ticket = queue + .pending_custody_stop([230; 16]) + .await + .ok_or("lost observer")?; + assert_eq!(queue.close_and_drain().await.len(), 1); + if fault == 0 { + assert!(matches!( + f.client().resolve(&invocation).await?, + Resolution::Absent + )); + edit( + &f, + "ALTER TABLE custody_stop_query_fault RENAME TO catalog_custody_commands", + ) + .await?; + } + if fault > 1 { + expired(&invocation).await?; + edit(&f, "UPDATE repository_identity SET owner='other'").await?; + assert!(matches!( + f.client().resolve(&invocation).await?, + Resolution::Expired + )); + } + queue.recover(&ticket).await?; + let outcome = stopped(observed(&ticket).await?)?; + assert_eq!(outcome.original, evidence); + assert_eq!(outcome.invocation, invocation); + assert_eq!( + outcome.committed.ok_or("stop result lost")?.output, + CustodyStopReply::Stopped + ); + assert!(outcome.stop.is_some()); + assert_eq!(queue.stats().await.command_bytes, 0); + assert!(queue.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +async fn junk(f: &Fixture, count: usize) -> Result { + edit(f, &format!("WITH RECURSIVE n(x) AS (VALUES(1) UNION ALL SELECT x+1 FROM n WHERE x<{count}) INSERT INTO catalog_custody_commands(purpose,operation,step,incarnation,request_id,intent) SELECT 0, CAST(printf('%016d',x) AS BLOB),0,zeroblob(16),CAST(printf('%016d',x) AS BLOB),x'01' FROM n")).await?; + Ok(()) +} + +#[tokio::test] +async fn stop_reclaims_only_pending_quota_and_retains_original_history() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let (evidence, _) = head_expiring(&f, 0, false, true).await?; + junk(&f, 1023).await?; + let next = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginPreparation(f.begin([248; 16])), + identity()?, + ) + .await?; + assert!(next.register(&f.client(), identity()?).await.is_err()); + expired(&evidence).await?; + let ready = registered(&f, 0) + .await? + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + ready.command_for_test().execute().await?; + next.register(&f.client(), identity()?).await?; + f.handle.query(0, 4096, |db| { + assert_eq!(db.query_row("SELECT count(*) FROM catalog_custody_commands WHERE phase IS NULL AND stopped IS NULL", [], |r| r.get::<_,i64>(0))?,1024); + assert_eq!(db.query_row("SELECT count(*) FROM catalog_custody_commands", [], |r| r.get::<_,i64>(0))?,1025); + Ok(Vec::new()) + }).await?; + assert_eq!(registered(&f, 0).await?.evidence(), &evidence); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn custody_scan_uses_bounded_indexed_pages_and_revisits_corruption_without_starving_tail_heads() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + junk(&f, 300).await?; + let (evidence, _) = head_expiring(&f, 0, false, true).await?; + expired(&evidence).await?; + f.handle.query(0,4096,|db| { + let mut query = db.prepare("EXPLAIN QUERY PLAN SELECT purpose,operation FROM catalog_custody_commands INDEXED BY catalog_custody_pending WHERE phase IS NULL AND stopped IS NULL AND (purpose,operation)>(?1,?2) ORDER BY purpose,operation LIMIT ?3")?; + let details: Vec = query.query_map(rusqlite::params![0,vec![0u8;16],17], |r| r.get(3))?.collect::>()?; + assert!(details.iter().any(|v|v.contains("SEARCH") && v.contains("catalog_custody_pending")),"{details:?}"); + assert!(details.iter().all(|v|!v.contains("TEMP B-TREE")),"{details:?}"); + Ok(Vec::new()) + }).await?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let limits = RecoveryScanLimits { + page: 17, + interval: Duration::from_millis(10), + }; + let service = CustodySupervisor::start( + f.client(), + f.target.clone(), + queue.clone(), + f.scans(limits), + f.authority(), + )?; + timeout(Duration::from_secs(10), async { + loop { + if service.stats().passes >= 2 && registered(&f, 0).await?.stop_fact().is_some() { + return Ok::<_, Box>(()); + } + tokio::task::yield_now().await; + } + }) + .await??; + let stats = service.stats(); + assert!(stats.failures >= 600); + assert!(stats.last_error.is_some()); + assert_eq!(stats.submitted, 1); + // A head behind the current cursor must be revisited on a later pass. + let next = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginStaging(f.begin([1; 16])), + identity()?, + ) + .await?; + next.register(&f.client(), identity()?).await?; + let passes = service.stats().passes; + timeout(Duration::from_secs(10), async { + while service.stats().passes < passes + 3 { + tokio::task::yield_now().await; + } + }) + .await?; + assert!(service.stats().deferred >= 1); + service.shutdown().await?; + assert!(queue.close_and_drain().await.is_empty()); + assert_eq!(queue.stats().await.command_bytes, 0); + for invalid in [0, 129] { + assert!( + CustodySupervisor::start( + f.client(), + f.target.clone(), + queue.clone(), + f.scans(RecoveryScanLimits { + page: invalid, + ..limits + }), + f.authority() + ) + .is_err() + ); + } + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn stopped_originals_survive_real_owner_restore_and_stop_sdk_expiry_without_current_permission() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 4, false, true).await?; + expired(&evidence).await?; + let original = registered(&f, 4).await?; + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let ready = original + .ready_stop(f.client(), mutation, &f.authority()) + .await?; + let invocation = ready.evidence().clone(); + let accepted = ready.command_for_test().execute().await?; + let expected = registered(&f, 4) + .await? + .stop_fact() + .ok_or("durable retirement")?; + assert_eq!(expected.receipt, accepted.receipt); + drop(ready); + drop(original); + edit(&f, "UPDATE repository_identity SET owner='other'").await?; + let (runtime, handle, client) = + super::durable_recovery::restore_owner_fence(&f, f.handle.owner_fence()).await?; + assert_ne!(handle.owner_fence(), expected.owner); + expired(&invocation).await?; + assert!(matches!( + client.resolve(&invocation).await?, + Resolution::Expired + )); + let loaded = RegisteredCustody::load_latest(&client, &f.target, [234; 16]) + .await? + .ok_or("restored original")?; + assert_eq!(loaded.evidence(), &evidence); + assert!(!loaded.settled()); + assert!(loaded.closed()); + assert_eq!(loaded.stop_fact(), Some(expected)); + assert!( + matches!(loaded.recover(&client).await,Err(InvocationError::Pending(value)) if *value==evidence) + ); + let stage = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let ticket = stage + .submit(ReadyStaging::restore(client, f.target.clone(), [234; 16]).await?) + .map_err(|(error, _)| error)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Fenced(_) + )); + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + assert!(ticket.restored_outcome().is_none()); + assert_eq!(stage.stats().command_bytes, 0); + assert!(stage.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn custody_scan_recovers_exact_maintenance_commands_after_their_pending_keys_disappear() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in 1..=3 { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 0, false, true).await?; + expired(&evidence).await?; + let staging = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let stage = staging + .submit(ReadyStaging::restore(f.client(), f.target.clone(), [230; 16]).await?) + .map_err(|(error, _)| error)?; + assert!(matches!( + timeout(Duration::from_secs(10), stage.wait_terminal()).await?, + StagingState::Uncertain(_) + )); + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + queue.fault_for_test(fault); + let service = CustodySupervisor::start( + f.client(), + f.target.clone(), + queue.clone(), + f.scans(RecoveryScanLimits { + page: 1, + interval: Duration::from_millis(100), + }), + f.authority(), + )?; + let invocation = timeout(Duration::from_secs(10), async { + loop { + if let Some(ticket) = queue.pending_custody_stop([230; 16]).await + && let PublicationState::Uncertain(error) = ticket.state() + && let PublicationError::CustodyStop(InvocationError::Pending(value)) = + &*error + { + return Ok::<_, Box>((**value).clone()); + } + tokio::task::yield_now().await; + } + }) + .await??; + timeout(Duration::from_secs(10), async { + loop { + if registered(&f, 0).await?.stop_fact().is_some() + && queue.stats().await.command_bytes == 0 + { + return Ok::<_, Box>(()); + } + tokio::task::yield_now().await; + } + }) + .await??; + let stats = service.shutdown().await?; + assert_eq!(stats.submitted, 1); + assert!(stats.recovered >= 1); + assert!(matches!( + f.client().resolve(&invocation).await?, + Resolution::Committed(_) + )); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + let fact = registered(&f, 0) + .await? + .stop_fact() + .ok_or("stopped original")?; + assert!(fact.receipt.commit_sequence > 0); + // No waiter or manual recovery is needed to observe logical closure. + timeout(Duration::from_secs(5), async { + while !matches!(stage.state(), StagingState::Fenced(_)) { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let StagingState::Fenced(error) = stage.state() else { + return Err("retired stage not fenced".into()); + }; + assert!( + matches!(&*error, StagingError::Custody { evidence: original, source } + if **original == evidence && matches!(&**source, CustodyError::Stopped(_))) + ); + assert_eq!(stage.restored_evidence(), Some(&evidence)); + assert!(stage.restored_outcome().is_none()); + assert_eq!(staging.stats().command_bytes, 0); + assert!(staging.close_and_drain().await.is_empty()); + assert!(queue.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn authenticated_stop_records_cannot_be_transplanted_to_another_original() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (evidence, _) = head_expiring(&f, 0, false, true).await?; + let (other, _) = head_expiring(&f, 4, false, true).await?; + expired(&evidence).await?; + expired(&other).await?; + let old = registered(&f, 0).await?; + for kind in [0, 4] { + registered(&f, kind) + .await? + .ready_stop(f.client(), identity()?, &f.authority()) + .await? + .command_for_test() + .execute() + .await?; + } + edit(&f, "DROP TRIGGER catalog_custody_stop_immutable").await?; + edit(&f, "UPDATE catalog_custody_commands SET stopped=(SELECT stopped FROM catalog_custody_commands WHERE operation=x'eaeaeaeaeaeaeaeaeaeaeaeaeaeaeaea') WHERE operation=x'e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6'").await?; + assert!( + RegisteredCustody::load_latest(&f.client(), &f.target, [230; 16]) + .await + .is_err() + ); + assert!( + matches!(old.recover(&f.client()).await, Err(InvocationError::Pending(value)) if *value==evidence) + ); + assert!( + ReadyStaging::restore(f.client(), f.target.clone(), [230; 16]) + .await + .is_err() + ); + assert!(registered(&f, 4).await?.stop_fact().is_some()); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn custody_stop_cannot_be_blocked_by_its_own_uncertain_preparation_and_fences_the_shared_session() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let operation = [230; 16]; + let begin = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginPreparation(f.begin(operation)), + identity()?, + ) + .await? + .register(&f.client(), identity()?) + .await? + .recover_preparation(&f.client()) + .await?; + let lease = lease(begin.output)?; + let session = Arc::new( + PreparationSession::open( + f.client(), + f.target.clone(), + check(lease.token), + Some(begin.receipt), + f.authority(), + ) + .await?, + ); + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let ready = session.ready_renew(mutation, DEFAULT_LEASE_MS).await?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + queue.fault_for_test(1); + let renewal = queue.submit(ready).await?; + assert!(matches!( + observed(&renewal).await?, + PublicationState::Uncertain(_) + )); + let original = registered(&f, 0).await?; + expired(original.evidence()).await?; + assert!(session.live_lease().is_ok()); + let service = CustodySupervisor::start( + f.client(), + f.target.clone(), + queue.clone(), + f.scans(RecoveryScanLimits { + page: 1, + interval: Duration::from_millis(10), + }), + f.authority(), + )?; + timeout(Duration::from_secs(10), async { + while queue.stats().await.command_bytes != 0 { + tokio::task::yield_now().await; + } + }) + .await?; + let stats = service.shutdown().await?; + assert_eq!(stats.submitted, 1); + let PublicationState::Finished(Err(error)) = renewal.state() else { + return Err("renewal not closed by its own stop".into()); + }; + assert!( + matches!(&*error, PublicationError::Custody { evidence, source } if **evidence==*original.evidence() && matches!(&**source,CustodyError::Stopped(_))) + ); + assert!(session.live_lease().is_err()); + assert!(registered(&f, 0).await?.closed()); + assert!(!registered(&f, 0).await?.settled()); + assert!(matches!( + f.client().resolve(original.evidence()).await?, + Resolution::Expired + )); + assert!(queue.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/durable_policy.rs b/crates/canopy-server/src/packs/publication/tests/durable_policy.rs index 0628d856..467200c7 100644 --- a/crates/canopy-server/src/packs/publication/tests/durable_policy.rs +++ b/crates/canopy-server/src/packs/publication/tests/durable_policy.rs @@ -68,13 +68,19 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write .await?; } else { edit(f, "CREATE TRIGGER phase_late_fault BEFORE UPDATE OF recovery_phase ON catalog_leases WHEN NEW.recovery_phase IS NOT NULL BEGIN SELECT RAISE(ABORT,'late phase fault'); END;").await?; - assert!(first.dispatch_any(&f.client(), store, &flag).await.is_err()); + assert!( + first + .dispatch_any(&f.client(), store, &f.authority(), &flag,) + .await + .is_err() + ); assert!(matches!( f.client().resolve(&first_evidence).await?, Resolution::Absent )); + let token = check.token; f.handle - .query(0, 128, |db| { + .query(0, 128, move |db| { assert_eq!( db.query_row("SELECT count(*) FROM ref_policy_guards", [], |row| row .get::<_, u64>(0))?, @@ -82,8 +88,8 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write ); assert_eq!( db.query_row( - "SELECT count(*) FROM catalog_leases WHERE recovery_phase IS NOT NULL", - [], + "SELECT count(*) FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND recovery_phase IS NOT NULL", + rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |row| row.get::<_, u64>(0) )?, 0 @@ -93,7 +99,9 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write .await?; edit(f, "DROP TRIGGER phase_late_fault").await?; } - let original = first.dispatch_any(&f.client(), store, &flag).await?; + let original = first + .dispatch_any(&f.client(), store, &f.authority(), &flag) + .await?; let mut head = first.clone(); let mut saved_first = None; let expected = if refusal_case { @@ -127,8 +135,9 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write let registered = page .persist_recovery(store, identity()?, Some(&head)) .await?; - let PublicationOutcome::PolicyPage(value) = - registered.dispatch_any(&f.client(), store, &flag).await? + let PublicationOutcome::PolicyPage(value) = registered + .dispatch_any(&f.client(), store, &f.authority(), &flag) + .await? else { return Err("successor page did not complete".into()); }; @@ -160,7 +169,7 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write let directory = root.to_path_buf(); let disk = budget.clone(); let ready = ticket - .spawn_bound(move |_| async move { + .spawn_bound(move |_, _context| async move { owner .ready_root_push(root_identity, &guard, &directory, disk, limits(), None) .await @@ -172,13 +181,14 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write head = ready .persist_recovery_after(store, identity()?, &head) .await?; - head.dispatch(&f.client(), store).await? + head.dispatch(&f.client(), store, &f.authority()).await? } }; // Return the same settled page through its retained predecessor frame. if let Some(expected) = &saved_first { - let PublicationOutcome::PolicyPage(actual) = - first.dispatch_any(&f.client(), store, &flag).await? + let PublicationOutcome::PolicyPage(actual) = first + .dispatch_any(&f.client(), store, &f.authority(), &flag) + .await? else { return Err("original page history missing".into()); }; @@ -218,7 +228,9 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write let loaded = RegisteredRootRecovery::load(&client, &f.target, store, &check) .await? .ok_or("restored durable phase")?; - let PublicationOutcome::RootPush(actual) = loaded.dispatch_any(&client, store, &flag).await? + let PublicationOutcome::RootPush(actual) = loaded + .dispatch_any(&client, store, &f.authority(), &flag) + .await? else { return Err("restored terminal phase missing".into()); }; @@ -247,8 +259,9 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write )) .await?; } - let PublicationOutcome::PolicyPage(actual) = - first.dispatch_any(&client, store, &flag).await? + let PublicationOutcome::PolicyPage(actual) = first + .dispatch_any(&client, store, &f.authority(), &flag) + .await? else { return Err("expired predecessor reply missing".into()); }; @@ -275,14 +288,24 @@ pub(super) async fn qualify(context: Context<'_>, refusal_case: bool, late_write Ok(Vec::new()) }) .await?; - Box::pin(query_failure(&loaded, &client, &handle, store, &expected)).await?; + Box::pin(query_failure( + &loaded, + &client, + &handle, + store, + &expected, + f.authority(), + f.publication_budget.clone(), + )) + .await?; let _released = Box::pin(super::terminal_retention::archive( f, &client, &handle, store, &loaded, &expected, 0, )) .await?; if let Some(expected) = &saved_first { - let PublicationOutcome::PolicyPage(actual) = - first.dispatch_any(&client, store, &flag).await? + let PublicationOutcome::PolicyPage(actual) = first + .dispatch_any(&client, store, &f.authority(), &flag) + .await? else { return Err("archived original page lost".into()); }; @@ -301,6 +324,8 @@ async fn query_failure( handle: &CellHandle, store: &canopy_object_storage::artifact::ArtifactStore, expected: &cellule_runtime::Committed, + authority: PreparationAuthority, + budget: PublicationBudget, ) -> Result { // The service is stopped and the original outcome has settled. Hide the // phase table to inject a real private-query failure without changing data. @@ -309,11 +334,14 @@ async fn query_failure( "ALTER TABLE catalog_leases RENAME TO phase_query_fault", ) .await?; - let ready = loaded.clone().ready(client.clone(), store.clone())?; + let ready = loaded + .clone() + .ready(client.clone(), store.clone(), authority)?; let reservation = ready.reservation(); let queue = PublicationCoordinator::new( loaded.evidence().target().clone(), PublicationLimits::default(), + budget, )?; let observer = queue .submit(ready) @@ -377,7 +405,7 @@ async fn late_write_case( // The lifecycle owns expensive preparation. Awaiting its typed result // keeps the original factory/custody checks without nesting the complete // native receive fixture on the producer's poll stack. - let worker = context.ticket.spawn_bound(move |_| async move { + let worker = context.ticket.spawn_bound(move |_, _context| async move { let publishing = owner .ready_root_push( positive_identity, @@ -422,7 +450,9 @@ async fn late_write_case( let registered = refusal .persist_recovery_after(store, identity()?, head) .await?; - let result = registered.dispatch(&f.client(), store).await?; + let result = registered + .dispatch(&f.client(), store, &f.authority()) + .await?; assert!( matches!(&result.output, RootCompletionReply::Completed(value) if value.completion.rejected && value.completion.publication.is_none()) diff --git a/crates/canopy-server/src/packs/publication/tests/durable_recovery.rs b/crates/canopy-server/src/packs/publication/tests/durable_recovery.rs index 59040b44..3fa9c7c9 100644 --- a/crates/canopy-server/src/packs/publication/tests/durable_recovery.rs +++ b/crates/canopy-server/src/packs/publication/tests/durable_recovery.rs @@ -154,12 +154,13 @@ async fn qualify_ready( .evidence(), &original ); + let token = check.token; let persisted = f .handle - .query(0, 1024, |db| { + .query(0, 1024, move |db| { Ok(db.query_row( - "SELECT recovery FROM catalog_leases WHERE recovery IS NOT NULL", - [], + "SELECT recovery FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND recovery IS NOT NULL", + rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |row| row.get::<_, Vec>(0), )?) }) @@ -187,7 +188,11 @@ async fn qualify_ready( .is_err() ); let original_result = if fault == 2 { - Some(registered.dispatch(&f.client(), store).await?) + Some( + registered + .dispatch(&f.client(), store, &f.authority()) + .await?, + ) } else { None }; @@ -218,17 +223,25 @@ async fn qualify_ready( .await? .ok_or("durable record missing after restore")?; assert_eq!(loaded.evidence(), &original); - let result = loaded.dispatch(&client, store).await; + let result = loaded.dispatch(&client, store, &f.authority()).await; if fault == 2 { let result = result?; assert!( matches!(&result.output, RootCompletionReply::Completed(value) if value.completion.publication.is_some() == publishing) ); let expected = original_result.ok_or("original outcome missing")?; - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; queue.fault_for_test(2); let observer = queue - .submit(loaded.clone().ready(client.clone(), store.clone())?) + .submit( + loaded + .clone() + .ready(client.clone(), store.clone(), f.authority())?, + ) .await .map_err(|failure| format!("durable admission: {:?}", failure.reason))?; assert!( @@ -250,7 +263,10 @@ async fn qualify_ready( (expected.output, expected.receipt) ); assert_eq!( - loaded.dispatch(&client, store).await?.receipt, + loaded + .dispatch(&client, store, &f.authority(),) + .await? + .receipt, result.receipt ); if revoked { @@ -267,7 +283,10 @@ async fn qualify_ready( )); // Known results still resolve; this grants no current response read. assert_eq!( - loaded.dispatch(&client, store).await?.receipt, + loaded + .dispatch(&client, store, &f.authority(),) + .await? + .receipt, result.receipt ); } else { @@ -289,7 +308,7 @@ async fn qualify_ready( .output .is_none() ); - assert_pin_retained(&handle).await?; + assert_pin_retained(&handle, check.token).await?; runtime.shutdown().await?; return Ok(()); } @@ -299,7 +318,9 @@ async fn qualify_ready( denied.output, RootCompletionReply::Denied(PreparationDenial::Stale) ); - assert!(matches!(loaded.dispatch(&client, store).await, + assert!(matches!(loaded.dispatch(&client, +store, +&f.authority(),).await, Err(PublicationError::RootPush(InvocationError::Rejected(replayed))) if replayed.receipt == denied.receipt && replayed.output == denied.output)); assert!( client @@ -309,7 +330,7 @@ async fn qualify_ready( .is_none() ); } - assert_pin_retained(&handle).await?; + assert_pin_retained(&handle, check.token).await?; runtime.shutdown().await?; Ok(()) } @@ -326,26 +347,26 @@ pub(super) async fn read_response( body, }) } -async fn assert_pin_retained(handle: &CellHandle) -> Result { +async fn assert_pin_retained(handle: &CellHandle, token: PreparationToken) -> Result { handle - .query(0, 32, |db| { + .query(0, 32, move |db| { assert_eq!( db.query_row( - "SELECT count(*) FROM catalog_leases WHERE recovery IS NOT NULL", - [], + "SELECT count(*) FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND recovery IS NOT NULL", + rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |row| row.get::<_, u64>(0) )?, 1 ); assert!( db.execute( - "UPDATE catalog_leases SET recovery=NULL WHERE recovery IS NOT NULL", - [] + "UPDATE catalog_leases SET recovery=NULL WHERE incarnation=?1 AND admission_sequence=?2 AND recovery IS NOT NULL", + rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt] ) .is_err() ); assert!( - db.execute("DELETE FROM catalog_leases WHERE recovery IS NOT NULL", []) + db.execute("DELETE FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND recovery IS NOT NULL", rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt]) .is_err() ); Ok(Vec::new()) @@ -357,6 +378,12 @@ async fn assert_pin_retained(handle: &CellHandle) -> Result { pub(super) async fn restore_owner( f: &Fixture, check: &LeaseCheck, +) -> Result<(CellRuntime, CellHandle, CellClient)> { + restore_owner_fence(f, check.token.owner).await +} +pub(super) async fn restore_owner_fence( + f: &Fixture, + old: OwnerFence, ) -> Result<(CellRuntime, CellHandle, CellClient)> { f.handle.drain().await?; f.runtime.shutdown().await?; @@ -394,7 +421,7 @@ pub(super) async fn restore_owner( }, ) .await?; - assert!(handle.owner_fence().epoch > check.token.owner.epoch); + assert!(handle.owner_fence().epoch > old.epoch); let client = CellClient::local(f.registry.clone(), handle.clone()); Ok((runtime, handle, client)) } diff --git a/crates/canopy-server/src/packs/publication/tests/initialization.rs b/crates/canopy-server/src/packs/publication/tests/initialization.rs index 8fff8765..87d75b42 100644 --- a/crates/canopy-server/src/packs/publication/tests/initialization.rs +++ b/crates/canopy-server/src/packs/publication/tests/initialization.rs @@ -10,7 +10,7 @@ use crate::packs::{ use canopy_object_storage::artifact::ArtifactStore; use cellule_ltx::DiskBudget; -async fn empty( +pub(super) async fn empty( fixture: &Fixture, operation: [u8; 16], store: Arc, @@ -30,17 +30,133 @@ fn initialized(reply: InitializationReply) -> Result { InitializationReply::Denied(why) => Err(format!("initialization denied {why:?}").into()), } } -async fn reject(fixture: &Fixture, input: InitialRefProof, reason: PreparationDenial) -> Result { - let before = state(&fixture.handle).await?; - let result = fixture +pub(super) async fn registered( + fixture: &Fixture, + prepared: &PreparedCatalog, + input: InitialRefProof, + mutation: MutationIdentity, +) -> Result<( + cellule_runtime::PreparedCommand, + RegisteredRootRecovery, +)> { + let command = fixture .client() - .command::(&fixture.target, identity()?, input) + .prepare_command::(&fixture.target, mutation, input) + .await?; + let record = super::super::recovery::persist( + &prepared.base.session, + &command, + super::super::recovery::Kind::Initialization, + &prepared.base.indexes().store(), + identity()?, + 0, + ) + .await?; + Ok((command, record)) +} +async fn reject( + fixture: &Fixture, + command: cellule_runtime::PreparedCommand, + registered: &RegisteredRootRecovery, + store: &ArtifactStore, + reason: PreparationDenial, +) -> Result { + let before = state(&fixture.handle).await?; + let evidence = command.evidence().clone(); + let result = command.execute().await?; + assert_eq!(result.output, InitializationReply::Denied(reason)); + assert_eq!(state(&fixture.handle).await?, before); + assert!(matches!( + fixture.client().resolve(&evidence).await?, + cellule_runtime::Resolution::Committed( + cellule_runtime::cell::executor::StoredOutcome::Success { .. } + ) + )); + let recovered = registered + .recover_initialization(&fixture.client(), store, &fixture.authority()) .await; assert!( - matches!(result,Err(InvocationError::Rejected(ref value)) if value.output==InitializationReply::Denied(reason)), + matches!(recovered,Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.receipt==result.receipt && value.output==result.output) + ); + Ok(()) +} + +#[tokio::test] +async fn unregistered_initialization_keeps_sdk_and_catalog_absent() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (prepared, root, budget) = empty(&fixture, [229; 16], store).await?; + let command = fixture + .client() + .prepare_command::( + &fixture.target, + identity()?, + prepared.empty_ref_initialization().await?, + ) + .await?; + let evidence = command.evidence().clone(); + let before = state(&fixture.handle).await?; + let result = command.execute().await; + assert!( + matches!(result, Err(InvocationError::NotStarted(_))), "{result:?}" ); assert_eq!(state(&fixture.handle).await?, before); + assert!(matches!( + fixture.client().resolve(&evidence).await?, + cellule_runtime::Resolution::Absent + )); + drop(prepared); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn cold_initialization_records_original_expiry_and_revocation_denials() -> Result { + for (sql, reason) in [ + ( + "UPDATE catalog_leases SET expires_at_ms=0; UPDATE catalog_operations SET expires_at_ms=0", + PreparationDenial::Expired, + ), + ( + "UPDATE repository_identity SET owner='replacement'", + PreparationDenial::Unauthorized, + ), + ] { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (prepared, root, budget) = empty(&fixture, [231; 16], store.clone()).await?; + let proof = prepared.empty_ref_initialization().await?; + let (command, registered) = registered(&fixture, &prepared, proof, identity()?).await?; + let evidence = command.evidence().clone(); + drop(command); + drop(prepared); + cleaned(root.path(), &budget).await?; + edit(&fixture, sql).await?; + let before = state(&fixture.handle).await?; + let result = registered + .recover_initialization(&fixture.client(), &store, &fixture.authority()) + .await; + assert!( + matches!(result,Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.output==InitializationReply::Denied(reason)), + "{result:?}" + ); + assert_eq!(state(&fixture.handle).await?, before); + assert!(matches!( + fixture.client().resolve(&evidence).await?, + cellule_runtime::Resolution::Committed( + cellule_runtime::cell::executor::StoredOutcome::Success { .. } + ) + )); + fixture.runtime.shutdown().await?; + } Ok(()) } @@ -78,10 +194,9 @@ async fn fresh_initialization_commits_joint_empty_roots_and_enables_first_ref_pr .is_none() ); let mutation = identity()?; - let committed = fixture - .client() - .command::(&fixture.target, mutation, proof.clone()) - .await?; + let (command, registered) = + registered(&fixture, &prepared, proof.clone(), mutation).await?; + let committed = command.clone().execute().await?; let fact = initialized(committed.output.clone())?; let mut e = BoundedEncoder::new(512)?; committed.output.encode(&mut e)?; @@ -130,13 +245,19 @@ async fn fresh_initialization_commits_joint_empty_roots_and_enables_first_ref_pr .receipt, committed.receipt ); - assert_eq!( + assert!(matches!( fixture .client() .command::(&fixture.target, identity()?, proof.clone()) - .await? - .output, - committed.output + .await, + Err(InvocationError::NotStarted(_)) + )); + let recovered = registered + .recover_initialization(&fixture.client(), &store, &fixture.authority()) + .await?; + assert_eq!( + (recovered.output, recovered.receipt), + (committed.output.clone(), committed.receipt) ); assert_eq!( fixture @@ -230,53 +351,76 @@ async fn initialization_refuses_history_head_changes_revocation_expiry_and_forge Arc::new(InMemory::new()), fixture.repository, )); - let (prepared, root, budget) = empty(&fixture, [223; 16], store).await?; + let (prepared, root, budget) = empty(&fixture, [223; 16], store.clone()).await?; let proof = prepared.empty_ref_initialization().await?; + let (command, registered) = registered(&fixture, &prepared, proof, identity()?).await?; edit(&fixture, sql).await?; - reject(&fixture, proof, reason).await?; + reject(&fixture, command, ®istered, &store, reason).await?; + drop(prepared); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + for mode in 0..3 { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (prepared, root, budget) = empty(&fixture, [224; 16], store.clone()).await?; + let proof = prepared.empty_ref_initialization().await?; + let mut forged = proof.clone(); + let mut data = proof.certificate.data()?; + if mode == 0 { + data.token.owner.epoch += 1; + } + if mode == 2 { + data.tenant = [96; 16]; + } + forged.certificate = + CatalogCertificate::seal(&data, &if mode == 1 { [17; 32] } else { [16; 32] })?; + let (command, registered) = registered(&fixture, &prepared, forged, identity()?).await?; + if mode == 0 { + let before = state(&fixture.handle).await?; + let evidence = command.evidence().clone(); + assert!(matches!( + command.execute().await, + Err(InvocationError::NotStarted(_)) + )); + assert_eq!(state(&fixture.handle).await?, before); + assert!(matches!( + fixture.client().resolve(&evidence).await?, + cellule_runtime::Resolution::Absent + )); + } else { + reject( + &fixture, + command, + ®istered, + &store, + PreparationDenial::Unauthorized, + ) + .await?; + } + let mut e = BoundedEncoder::new(128)?; + proof.refs.encode(&mut e)?; + let mut bytes = e.finish(); + let end = bytes.len() - 1; + bytes[end] ^= 1; + let mut d = BoundedDecoder::new(&bytes, 128)?; + let refs = RefStateSnapshotRoot::decode(&mut d)?; + d.finish()?; + let raw = InitialRefProof { + refs, + certificate: proof.certificate, + }; + assert!( + raw.encode(&mut BoundedEncoder::new(INITIALIZATION_BYTES)?) + .is_err() + ); drop(prepared); cleaned(root.path(), &budget).await?; fixture.runtime.shutdown().await?; } - let fixture = Fixture::new(ObjectFormat::Sha256).await?; - let store = Arc::new(ArtifactStore::new( - Arc::new(InMemory::new()), - fixture.repository, - )); - let (prepared, root, budget) = empty(&fixture, [224; 16], store).await?; - let proof = prepared.empty_ref_initialization().await?; - let mut stale = proof.clone(); - let mut data = stale.certificate.data()?; - data.token.owner.epoch += 1; - stale.certificate = CatalogCertificate::seal(&data, &[16; 32])?; - reject(&fixture, stale, PreparationDenial::Stale).await?; - let mut forged = proof.clone(); - forged.certificate = CatalogCertificate::seal(&proof.certificate.data()?, &[17; 32])?; - reject(&fixture, forged, PreparationDenial::Unauthorized).await?; - let mut wrong = proof.clone(); - let mut data = wrong.certificate.data()?; - data.tenant = [96; 16]; - wrong.certificate = CatalogCertificate::seal(&data, &[16; 32])?; - reject(&fixture, wrong, PreparationDenial::Unauthorized).await?; - let mut e = BoundedEncoder::new(128)?; - proof.refs.encode(&mut e)?; - let mut bytes = e.finish(); - let end = bytes.len() - 1; - bytes[end] ^= 1; - let mut d = BoundedDecoder::new(&bytes, 128)?; - let refs = RefStateSnapshotRoot::decode(&mut d)?; - d.finish()?; - let raw = InitialRefProof { - refs, - certificate: proof.certificate, - }; - assert!( - raw.encode(&mut BoundedEncoder::new(INITIALIZATION_BYTES)?) - .is_err() - ); - drop(prepared); - cleaned(root.path(), &budget).await?; - fixture.runtime.shutdown().await?; Ok(()) } @@ -292,16 +436,34 @@ async fn initialization_late_failure_rolls_back_roots_checkpoint_and_outcome_and let (second, root_b, budget_b) = empty(&fixture, [226; 16], store).await?; let a = first.empty_ref_initialization().await?; let b = second.empty_ref_initialization().await?; + let (command_a, registered_a) = registered(&fixture, &first, a, identity()?).await?; + let (command_b, registered_b) = registered(&fixture, &second, b, identity()?).await?; edit(&fixture,"CREATE TRIGGER fail_initialization BEFORE INSERT ON catalog_initialization BEGIN SELECT RAISE(ABORT,'late initialization fault'); END;").await?; let before = state(&fixture.handle).await?; + let failed = command_a.clone().execute().await; assert!( - fixture - .client() - .command::(&fixture.target, identity()?, a.clone()) - .await - .is_err() + matches!(failed,Err(InvocationError::NotStarted(Error::Sqlite(rusqlite::Error::SqliteFailure(_,Some(ref message))))) if message == "late initialization fault"), + "{failed:?}" ); assert_eq!(state(&fixture.handle).await?, before); + fixture + .handle + .query(0, 128, |db| { + assert_eq!( + db.query_row( + "SELECT count(*) FROM catalog_leases WHERE recovery_phase IS NOT NULL", + [], + |row| row.get::<_, u64>(0) + )?, + 0 + ); + Ok(Vec::new()) + }) + .await?; + assert!(matches!( + fixture.client().resolve(command_a.evidence()).await?, + cellule_runtime::Resolution::Absent + )); assert!( fixture .client() @@ -312,29 +474,25 @@ async fn initialization_late_failure_rolls_back_roots_checkpoint_and_outcome_and ); edit(&fixture, "DROP TRIGGER fail_initialization").await?; let client = fixture.client(); - let (result_a, result_b) = tokio::join!( - client.command::(&fixture.target, identity()?, a.clone()), - client.command::(&fixture.target, identity()?, b.clone()), - ); - let (committed, losing, winner, loser) = match (result_a, result_b) { - (Ok(committed), Err(InvocationError::Rejected(rejected))) => { - assert_eq!( - rejected.output, - InitializationReply::Denied(PreparationDenial::Conflict) - ); - (committed, b, [225; 16], [226; 16]) - } - (Err(InvocationError::Rejected(rejected)), Ok(committed)) => { - assert_eq!( - rejected.output, - InitializationReply::Denied(PreparationDenial::Conflict) - ); - (committed, a, [226; 16], [225; 16]) - } - other => { - return Err(format!("initialization must have exactly one winner: {other:?}").into()); - } - }; + let (result_a, result_b) = tokio::join!(command_a.execute(), command_b.execute()); + let result_a = result_a?; + let result_b = result_b?; + let (committed, losing, winner, loser, registered_loser) = + match (&result_a.output, &result_b.output) { + ( + InitializationReply::Initialized(_), + InitializationReply::Denied(PreparationDenial::Conflict), + ) => (result_a, result_b, [225; 16], [226; 16], registered_b), + ( + InitializationReply::Denied(PreparationDenial::Conflict), + InitializationReply::Initialized(_), + ) => (result_b, result_a, [226; 16], [225; 16], registered_a), + other => { + return Err( + format!("initialization must have exactly one winner: {other:?}").into(), + ); + } + }; let fact = initialized(committed.output)?; assert_eq!(fact.generation, 1); assert_eq!( @@ -351,7 +509,12 @@ async fn initialization_late_failure_rolls_back_roots_checkpoint_and_outcome_and .output .is_none() ); - reject(&fixture, losing, PreparationDenial::Conflict).await?; + let recovered = registered_loser + .recover_initialization(&client, &first.base.indexes().store(), &fixture.authority()) + .await; + assert!( + matches!(recovered,Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.receipt==losing.receipt && value.output==losing.output) + ); drop(first); drop(second); cleaned(root_a.path(), &budget_a).await?; @@ -373,10 +536,11 @@ async fn initialization_exact_outcome_survives_owner_restore_and_pending_old_att let a = first.empty_ref_initialization().await?; let b = second.empty_ref_initialization().await?; let mutation = identity()?; - let committed = fixture - .client() - .command::(&fixture.target, mutation, a.clone()) - .await?; + let (command_a, registered_a) = registered(&fixture, &first, a.clone(), mutation).await?; + let (command_b, registered_b) = registered(&fixture, &second, b.clone(), identity()?).await?; + let snapshot_b = command_b.snapshot(); + let body_b = command_b.input_bytes().to_vec(); + let committed = command_a.execute().await?; fixture.handle.drain().await?; fixture.runtime.shutdown().await?; let session = SessionId::from_bytes([229; 16]); @@ -412,21 +576,45 @@ async fn initialization_exact_outcome_survives_owner_restore_and_pending_old_att (exact.output, exact.receipt), (committed.output.clone(), committed.receipt) ); - assert_eq!( + assert!(matches!( client .command::(&fixture.target, identity()?, a) - .await? - .output, - committed.output + .await, + Err(InvocationError::NotStarted(_)) + )); + let recovered = registered_a + .recover_initialization(&client, &first.base.indexes().store(), &fixture.authority()) + .await?; + assert_eq!( + (recovered.output, recovered.receipt), + (committed.output.clone(), committed.receipt) ); let before = state(&handle).await?; - let stale = client - .command::(&fixture.target, identity()?, b) + // Cold recovery must settle the absent original under the new owner. It + // cannot depend on opening the old owner's now-invalid live capability. + let denied = registered_b + .recover_initialization( + &client, + &second.base.indexes().store(), + &fixture.authority(), + ) .await; assert!( - matches!(stale,Err(InvocationError::Rejected(ref value)) if value.output==InitializationReply::Denied(PreparationDenial::Stale)) + matches!(denied,Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.output==InitializationReply::Denied(PreparationDenial::Stale)), + "{denied:?}" + ); + let stale = client + .restore_command::(snapshot_b, body_b)? + .execute() + .await?; + assert_eq!( + stale.output, + InitializationReply::Denied(PreparationDenial::Stale) ); assert_eq!(state(&handle).await?, before); + assert!( + matches!(denied,Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.receipt==stale.receipt && value.output==stale.output) + ); assert_eq!( client .query::(&fixture.target, None, fixture.begin([227; 16])) diff --git a/crates/canopy-server/src/packs/publication/tests/initialization_recovery.rs b/crates/canopy-server/src/packs/publication/tests/initialization_recovery.rs new file mode 100644 index 00000000..21b0c942 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/initialization_recovery.rs @@ -0,0 +1,312 @@ +//! Original initialization identity survives real restore and SDK expiry. +use super::*; +use super::{ + initialization::empty, + prepare::cleaned, + publishing::{edit, state}, +}; +use crate::packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}, + metadata::tests::limits, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use cellule_runtime::Resolution; +use object_store::{ObjectStore, ObjectStoreExt}; +use tokio::time::{Duration, timeout}; + +#[tokio::test] +async fn lost_initialization_registration_is_discovered_after_fresh_disk_owner_restore() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let operation = [233; 16]; + let (prepared, root, budget) = empty(&f, operation, store.clone()).await?; + let command = f + .client() + .prepare_command::( + &f.target, + identity()?, + prepared.empty_ref_initialization().await?, + ) + .await?; + let original = command.evidence().clone(); + let check = check(prepared.token()); + let lost = super::super::recovery::persist( + &prepared.base.session, + &command, + super::super::recovery::Kind::Initialization, + &store, + identity()?, + 2, + ) + .await; + assert!( + matches!(lost, Err(RootRecoveryError::Registration(ref error)) if matches!(&**error, InvocationError::Pending(_))) + ); + assert!(matches!( + f.client().resolve(&original).await?, + Resolution::Absent + )); + let saved = RegisteredRootRecovery::load_initialization( + &f.client(), + &f.target, + &store, + &f.begin(operation), + ) + .await? + .ok_or("winning initialization registration absent")?; + assert_eq!(saved.evidence(), &original); + let rival = f + .client() + .prepare_command::( + &f.target, + identity()?, + prepared.empty_ref_initialization().await?, + ) + .await?; + super::mandatory_registration::not_started(&f, &rival).await?; + let rival_registration = super::super::recovery::persist( + &prepared.base.session, + &rival, + super::super::recovery::Kind::Initialization, + &store, + identity()?, + 0, + ) + .await; + assert!( + matches!(rival_registration, Err(RootRecoveryError::Registration(ref error)) if matches!(&**error, InvocationError::Rejected(value) if value.output==RootRecoveryReply::Denied(PreparationDenial::Conflict))) + ); + let mut wrong = f.begin(operation); + wrong.request_digest[0] ^= 1; + assert!( + RegisteredRootRecovery::load_initialization(&f.client(), &f.target, &store, &wrong) + .await? + .is_none() + ); + drop(rival); + drop(command); + drop(saved); + drop(prepared); + cleaned(root.path(), &budget).await?; + let (runtime, handle, client) = super::durable_recovery::restore_owner(&f, &check).await?; + assert!(!f.root.path().join("a.sqlite").exists()); + let saved = RegisteredRootRecovery::load_initialization( + &client, + &f.target, + &store, + &f.begin(operation), + ) + .await? + .ok_or("restored registration absent")?; + assert_eq!(saved.evidence(), &original); + assert!(matches!( + client.resolve(&original).await?, + Resolution::Absent + )); + let before = state(&handle).await?; + let denied = match saved + .recover_initialization(&client, &store, &f.authority()) + .await + { + Err(PublicationError::Initialization(InvocationError::Rejected(value))) => value, + other => { + return Err(format!("cold original must settle its stale owner: {other:?}").into()); + } + }; + assert_eq!( + denied.output, + InitializationReply::Denied(PreparationDenial::Stale) + ); + assert_eq!(state(&handle).await?, before); + assert!(matches!(saved.recover_initialization(&client, +&store, +&f.authority(),).await, Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.receipt==denied.receipt)); + // Only a definitive original denial permits a new owner to claim. It gets + // its own namespace; the original pin and result remain unchanged. + let started = client + .command::( + &f.target, + identity()?, + LeaseRequest { + check: check.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + let current = lease(started.output)?; + assert_eq!(current.token.owner, handle.owner_fence()); + assert_ne!( + current.token.artifact_operation, + check.token.artifact_operation + ); + assert_eq!(current.base.generation, 0); + let scratch = tempfile::TempDir::new()?; + let disk = DiskBudget::new(64 << 20); + let indexes = Arc::new(CatalogIndexes::new(store.clone(), f.format)); + let files = Arc::new(CatalogFiles::new( + scratch.path(), + disk.clone(), + store.clone(), + f.format, + CatalogFileLimits::default(), + )?); + let base = Arc::new( + PreparationBaseResolver::open( + client.clone(), + f.target.clone(), + super::check(current.token), + indexes, + files, + Some(started.receipt), + f.authority(), + ) + .await?, + ); + let prepared = Arc::new( + CatalogPreparation::new(scratch.path(), disk.clone(), base, limits()) + .await? + .finish() + .await?, + ); + let ready = prepared.ready_initialization(identity()?).await?; + let registered = ready.persist_recovery(&store, identity()?).await?; + let result = ready.complete(®istered, &store).await?; + assert!( + matches!(result.output, InitializationReply::Initialized(ref fact) if fact.generation==1) + ); + assert!(matches!(saved.recover_initialization(&client, +&store, +&f.authority(),).await, Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.receipt==denied.receipt)); + handle + .query(0, 128, |db| { + assert_eq!( + db.query_row( + "SELECT artifact_sequence FROM repository_identity", + [], + |r| r.get::<_, u64>(0) + )?, + 2 + ); + assert_eq!( + db.query_row("SELECT count(*) FROM catalog_initialization", [], |r| r + .get::<_, u64>(0))?, + 1 + ); + assert_eq!( + db.query_row( + "SELECT count(*) FROM catalog_leases WHERE recovery_phase IS NOT NULL", + [], + |r| r.get::<_, u64>(0) + )?, + 2 + ); + Ok(Vec::new()) + }) + .await?; + drop(prepared); + cleaned(scratch.path(), &disk).await?; + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn original_initialization_receipt_survives_lost_ack_expiry_body_loss_and_owner_restore() +-> Result { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + let (prepared, root, budget) = empty(&f, [234; 16], store.clone()).await?; + let prepared = Arc::new(prepared); + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 8_000; + let ready = prepared.ready_initialization(mutation).await?; + let registered = ready.persist_recovery(&store, identity()?).await?; + let original = registered.evidence().clone(); + let check = check(registered.token()); + let bound = ready.bind_recovery(registered.clone(), &store)?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + queue.fault_for_test(2); + let observer = queue + .submit(bound) + .await + .map_err(|failure| format!("initialization admission: {:?}", failure.reason))?; + assert!( + matches!(timeout(Duration::from_secs(10), observer.wait()).await?, PublicationState::Uncertain(ref error) if matches!(&**error, PublicationError::Initialization(InvocationError::Pending(evidence)) if **evidence==original)) + ); + assert_eq!(queue.stats().await.command_bytes, 20 << 10); + assert_eq!(queue.close_and_drain().await.len(), 1); + observer.recover().await?; + let expected = match timeout(Duration::from_secs(10), observer.wait()).await? { + PublicationState::Finished(Ok(PublicationOutcome::Initialization(value))) => value, + other => return Err(format!("original initialization recovery: {other:?}").into()), + }; + assert!(matches!( + expected.output, + InitializationReply::Initialized(_) + )); + assert_eq!(queue.stats().await.command_bytes, 0); + assert_eq!(queue.stats().await.admitted, 0); + assert!(queue.close_and_drain().await.is_empty()); + drop(observer); + drop(queue); + drop(prepared); + cleaned(root.path(), &budget).await?; + // Delete the exact saved body manifest and revoke current rights. Original + // phase knowledge must win before both body I/O and fresh permission checks. + for (key, descriptor) in registered.command_bodies_for_test() { + let path = store.path(key, descriptor.digest)?; + provider.head(&path).await?; + provider.delete(&path).await?; + assert!(matches!( + provider.head(&path).await, + Err(object_store::Error::NotFound { .. }) + )); + } + edit(&f, "UPDATE repository_identity SET owner='replacement'").await?; + drop(registered); + let (runtime, handle, client) = super::durable_recovery::restore_owner(&f, &check).await?; + let now = i64::try_from(SystemTime::now().duration_since(UNIX_EPOCH)?.as_millis())?; + if now <= mutation.expires_at_ms { + tokio::time::sleep(Duration::from_millis(u64::try_from( + mutation.expires_at_ms - now + 1, + )?)) + .await; + } + assert!(matches!( + client.resolve(&original).await?, + Resolution::Expired + )); + let loaded = RegisteredRootRecovery::load(&client, &f.target, &store, &check) + .await? + .ok_or("original initialization pin absent")?; + assert_eq!(loaded.evidence(), &original); + let before = state(&handle).await?; + let result = loaded + .recover_initialization(&client, &store, &f.authority()) + .await?; + assert_eq!( + (result.output, result.receipt), + (expected.output.clone(), expected.receipt) + ); + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let observer = queue + .submit(loaded.ready(client, (*store).clone(), f.authority())?) + .await + .map_err(|failure| format!("cold initialization admission: {:?}", failure.reason))?; + assert!( + matches!(timeout(Duration::from_secs(10), observer.wait()).await?, PublicationState::Finished(Ok(PublicationOutcome::Initialization(ref value))) if value.receipt==expected.receipt && value.output==expected.output) + ); + assert!(queue.close_and_drain().await.is_empty()); + assert_eq!(state(&handle).await?, before); + runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/initialization_retirement.rs b/crates/canopy-server/src/packs/publication/tests/initialization_retirement.rs new file mode 100644 index 00000000..babd4b7b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/initialization_retirement.rs @@ -0,0 +1,529 @@ +//! Release the initial floor without losing the exact original command result. +use super::*; +use super::{ + initialization::empty, + prepare::cleaned, + publishing::{edit, edit_handle}, + terminal_retention::maintenance, +}; +use crate::packs::catalog::CatalogSnapshot; +use canopy_object_storage::artifact::{ArtifactKey, ArtifactKind, ArtifactStore}; +use cellule_runtime::Resolution; +use object_store::{ObjectStore, ObjectStoreExt}; +use tokio::time::{Duration, timeout}; + +#[tokio::test] +async fn initialized_repository_releases_zero_floor_and_recovers_original_receipt() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let (prepared, root, budget) = empty(&f, [235; 16], store.clone()).await?; + let prepared = Arc::new(prepared); + let ready = prepared.ready_initialization(identity()?).await?; + let registered = ready.persist_recovery(&store, identity()?).await?; + let admin = maintenance(&f.handle, f.repository).await?; + assert!( + registered + .ready_terminal_release(f.client(), &store, admin.clone(), identity()?) + .await + .is_err() + ); + let original = ready.complete(®istered, &store).await?; + let discovered = RegisteredRootRecovery::load_initialization( + &f.client(), + &f.target, + &store, + &f.begin([235; 16]), + ) + .await? + .ok_or("closed initial pin not discovered")?; + assert_eq!(discovered.evidence(), registered.evidence()); + let result = registered + .ready_terminal_release(f.client(), &store, admin, identity()?) + .await? + .complete() + .await?; + assert_eq!(result.output, TerminalReleaseReply::Released); + f.handle.query(0, 128, |db| { + assert_eq!(db.query_row("SELECT count(*) FROM catalog_leases", [], |r| r.get::<_,u64>(0))?, 0); + assert_eq!(db.query_row("SELECT count(*) FROM catalog_initialization", [], |r| r.get::<_,u64>(0))?, 1); + assert_eq!(db.query_row("SELECT count(*) FROM catalog_recovery_receipts", [], |r| r.get::<_,u64>(0))?, 1); + for sql in [ + "UPDATE catalog_recovery_receipts SET operation=zeroblob(16)", + "UPDATE catalog_recovery_receipts SET recovery_phase=x'01'", + "UPDATE catalog_recovery_receipts SET recovery_release=x'01'", + "INSERT OR REPLACE INTO catalog_recovery_receipts SELECT * FROM catalog_recovery_receipts", + "DELETE FROM catalog_recovery_receipts", + ] { + assert!(db.execute(sql, []).is_err(), "{sql}"); + } + Ok(Vec::new()) + }).await?; + let check = check(registered.token()); + drop(discovered); + drop(registered); + drop(prepared); + cleaned(root.path(), &budget).await?; + let loaded = RegisteredRootRecovery::load(&f.client(), &f.target, &store, &check) + .await? + .ok_or("initial receipt archive missing")?; + let recovered = loaded + .recover_initialization(&f.client(), &store, &f.authority()) + .await?; + assert_eq!( + (recovered.output, recovered.receipt), + (original.output, original.receipt) + ); + let mut wrong = check; + wrong.token.attempt += 1; + assert!( + RegisteredRootRecovery::load(&f.client(), &f.target, &store, &wrong) + .await? + .is_none() + ); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn initialization_retirement_checks_actual_authority_and_rolls_back_the_last_write() -> Result +{ + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let (prepared, root, budget) = empty(&f, [236; 16], store.clone()).await?; + let prepared = Arc::new(prepared); + let ready = prepared.ready_initialization(identity()?).await?; + let saved = ready.persist_recovery(&store, identity()?).await?; + ready.complete(&saved, &store).await?; + let admin = maintenance(&f.handle, f.repository).await?; + for wrong_owner in [true, false] { + let mut wrong = admin.clone(); + if wrong_owner { + wrong.owner.epoch += 1; + } else { + wrong.actor = "outsider".into(); + } + let release = saved + .ready_terminal_release(f.client(), &store, wrong, identity()?) + .await?; + assert!( + matches!(release.complete().await, Err(PublicationError::TerminalRelease(InvocationError::Rejected(ref value))) if value.output == TerminalReleaseReply::Denied(PreparationDenial::Unauthorized)) + ); + } + let release = saved + .ready_terminal_release(f.client(), &store, admin, identity()?) + .await?; + edit_handle(&f.handle, "CREATE TRIGGER initialization_release_late_fault BEFORE DELETE ON catalog_leases WHEN OLD.recovery IS NOT NULL BEGIN SELECT RAISE(ABORT,'late initialization release fault'); END").await?; + let failed = release.clone().complete().await; + assert!( + matches!(failed, Err(PublicationError::TerminalRelease(InvocationError::NotStarted(Error::Sqlite(rusqlite::Error::SqliteFailure(_,Some(ref message)))))) if message == "late initialization release fault"), + "{failed:?}" + ); + assert!(matches!( + f.client().resolve(&release.evidence_for_test()).await?, + Resolution::Absent + )); + f.handle.query(0,128,|db| { + assert_eq!(db.query_row("SELECT count(*) FROM catalog_recovery_receipts",[],|r| r.get::<_,u64>(0))?,0); + assert_eq!(db.query_row("SELECT count(*) FROM catalog_leases WHERE recovery IS NOT NULL AND recovery_phase IS NOT NULL",[],|r| r.get::<_,u64>(0))?,1); + Ok(Vec::new()) + }).await?; + edit_handle(&f.handle, "DROP TRIGGER initialization_release_late_fault").await?; + assert_eq!( + release.complete().await?.output, + TerminalReleaseReply::Released + ); + drop(prepared); + cleaned(root.path(), &budget).await?; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn missing_typed_initial_metadata_cannot_authorize_retirement() -> Result { + for missing in 0..3 { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let provider = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + let (prepared, root, budget) = empty(&f, [237; 16], store.clone()).await?; + let prepared = Arc::new(prepared); + let ready = prepared.ready_initialization(identity()?).await?; + let saved = ready.persist_recovery(&store, identity()?).await?; + let original = ready.complete(&saved, &store).await?; + let InitializationReply::Initialized(fact) = &original.output else { + return Err("initialization reply".into()); + }; + let catalog = fact.catalog.ok_or("catalog absent")?; + let directory = CatalogSnapshot::download(&store, catalog).await?.directory; + let refs = fact.refs.ok_or("refs absent")?; + let (operation, kind, artifact) = match missing { + 0 => ( + catalog.operation, + ArtifactKind::CatalogNode, + catalog.artifact, + ), + 1 => ( + directory.operation, + ArtifactKind::CatalogNode, + directory.artifact, + ), + _ => (refs.operation(), ArtifactKind::InputRoot, refs.artifact()), + }; + let path = store.path( + ArtifactKey { + operation, + binding_digest: artifact.digest, + kind, + }, + artifact.digest, + )?; + provider.head(&path).await?; + provider.delete(&path).await?; + assert!( + saved + .ready_terminal_release( + f.client(), + &store, + maintenance(&f.handle, f.repository).await?, + identity()? + ) + .await + .is_err() + ); + assert_eq!( + saved + .recover_initialization(&f.client(), &store, &f.authority(),) + .await? + .receipt, + original.receipt + ); + f.handle + .query(0, 128, |db| { + assert_eq!( + db.query_row( + "SELECT count(*) FROM catalog_leases WHERE recovery IS NOT NULL", + [], + |r| r.get::<_, u64>(0) + )?, + 1 + ); + assert_eq!( + db.query_row("SELECT count(*) FROM catalog_recovery_receipts", [], |r| { + r.get::<_, u64>(0) + })?, + 0 + ); + Ok(Vec::new()) + }) + .await?; + drop(prepared); + cleaned(root.path(), &budget).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn denied_initial_attempt_retires_only_after_claim_and_keeps_its_receipt_after_success() +-> Result { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let operation = [238; 16]; + let (prepared, root, budget) = empty(&f, operation, store.clone()).await?; + let prepared = Arc::new(prepared); + let ready = prepared.ready_initialization(identity()?).await?; + let saved = ready.persist_recovery(&store, identity()?).await?; + let old = check(saved.token()); + edit( + &f, + "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0", + ) + .await?; + let denied = match saved + .recover_initialization(&f.client(), &store, &f.authority()) + .await + { + Err(PublicationError::Initialization(InvocationError::Rejected(value))) => value, + other => return Err(format!("expected original expiry: {other:?}").into()), + }; + assert_eq!( + denied.output, + InitializationReply::Denied(PreparationDenial::Expired) + ); + let admin = maintenance(&f.handle, f.repository).await?; + assert!( + saved + .ready_terminal_release(f.client(), &store, admin.clone(), identity()?) + .await + .is_err() + ); + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let supervisor = RecoverySupervisor::start_retiring( + f.client(), + f.target.clone(), + (*store).clone(), + queue.clone(), + f.scans(RecoveryScanLimits { + page: 1, + interval: Duration::from_secs(1), + }), + f.authority(), + admin.clone(), + )?; + timeout(Duration::from_secs(10), async { + while supervisor.stats().scanned == 0 { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let scan = supervisor.shutdown().await?; + assert_eq!(scan.release_submitted, 0); + assert!(scan.deferred > 0); + assert_eq!(queue.stats().await.admitted, 0); + f.client() + .command::( + &f.target, + identity()?, + LeaseRequest { + check: old.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + let release = saved + .ready_terminal_release(f.client(), &store, admin.clone(), identity()?) + .await?; + release.complete().await?; + drop(ready); + drop(prepared); + cleaned(root.path(), &budget).await?; + let (prepared, root, budget) = empty(&f, operation, store.clone()).await?; + let prepared = Arc::new(prepared); + let ready = prepared.ready_initialization(identity()?).await?; + let current = ready.persist_recovery(&store, identity()?).await?; + assert_ne!( + current.token().artifact_operation, + saved.token().artifact_operation + ); + assert_eq!( + RegisteredRootRecovery::load_initialization( + &f.client(), + &f.target, + &store, + &f.begin(operation) + ) + .await? + .ok_or("new attempt absent")? + .evidence(), + current.evidence() + ); + ready.complete(¤t, &store).await?; + current + .ready_terminal_release(f.client(), &store, admin, identity()?) + .await? + .complete() + .await?; + let old = RegisteredRootRecovery::load(&f.client(), &f.target, &store, &old) + .await? + .ok_or("old denied archive absent")?; + assert!(matches!(old.recover_initialization(&f.client(), +&store, +&f.authority(),).await, Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) if value.output == denied.output && value.receipt == denied.receipt)); + f.handle + .query(0, 128, |db| { + assert_eq!( + db.query_row("SELECT count(*) FROM catalog_recovery_receipts", [], |r| { + r.get::<_, u64>(0) + })?, + 2 + ); + assert_eq!( + db.query_row("SELECT count(*) FROM catalog_leases", [], |r| r + .get::<_, u64>(0))?, + 0 + ); + Ok(Vec::new()) + }) + .await?; + assert!(queue.close_and_drain().await.is_empty()); + drop(prepared); + cleaned(root.path(), &budget).await?; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn lost_initial_retirement_ack_keeps_original_receipts_after_expiry_body_loss_and_restore() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + let (prepared, root, budget) = empty(&f, [239; 16], store.clone()).await?; + let prepared = Arc::new(prepared); + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 8_000; + let ready = prepared.ready_initialization(mutation).await?; + let saved = ready.persist_recovery(&store, identity()?).await?; + let original = ready.complete(&saved, &store).await?; + let check = check(saved.token()); + let mut release_identity = identity()?; + release_identity.expires_at_ms = release_identity.issued_at_ms + 8_000; + let release = saved + .ready_terminal_release( + f.client(), + &store, + maintenance(&f.handle, f.repository).await?, + release_identity, + ) + .await?; + let rival = saved + .ready_terminal_release( + f.client(), + &store, + maintenance(&f.handle, f.repository).await?, + identity()?, + ) + .await?; + let evidence = release.evidence_for_test(); + assert!(matches!( + release.clone().dispatch(false, 2).await, + Err(PublicationError::TerminalRelease(InvocationError::Pending( + _ + ))) + )); + let released = release.clone().complete().await?; + assert!( + matches!(rival.complete().await, Err(PublicationError::TerminalRelease(InvocationError::Rejected(ref value))) if value.output == TerminalReleaseReply::Denied(PreparationDenial::Missing)) + ); + for (key, descriptor) in saved.command_bodies_for_test() { + let path = store.path(key, descriptor.digest)?; + provider.head(&path).await?; + provider.delete(&path).await?; + } + drop(prepared); + cleaned(root.path(), &budget).await?; + edit(&f, "UPDATE repository_identity SET owner='replacement'").await?; + let (runtime, handle, client) = super::durable_recovery::restore_owner(&f, &check).await?; + assert_ne!(handle.owner_fence(), check.token.owner); + loop { + let now = i64::try_from(SystemTime::now().duration_since(UNIX_EPOCH)?.as_millis())?; + if now > release_identity.expires_at_ms.max(mutation.expires_at_ms) { + break; + } + tokio::time::sleep(Duration::from_millis(u64::try_from( + release_identity.expires_at_ms.max(mutation.expires_at_ms) - now + 1, + )?)) + .await; + } + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Expired + )); + let restored = RegisteredRootRecovery::load(&client, &f.target, &store, &check) + .await? + .ok_or("restored archive absent")?; + let result = restored + .recover_initialization(&client, &store, &f.authority()) + .await?; + assert_eq!( + (result.output, result.receipt), + (original.output, original.receipt) + ); + let result = release.with_client_for_test(client).complete().await?; + assert_eq!( + (result.output, result.receipt), + (released.output, released.receipt) + ); + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn automatic_initialization_retirement_recovers_uncertainty_after_pin_disappears() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let (prepared, root, budget) = empty(&f, [240; 16], store.clone()).await?; + let prepared = Arc::new(prepared); + let ready = prepared.ready_initialization(identity()?).await?; + let saved = ready.persist_recovery(&store, identity()?).await?; + ready.complete(&saved, &store).await?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + queue.fault_for_test(2); + let scanner = RecoverySupervisor::start_retiring( + f.client(), + f.target.clone(), + (*store).clone(), + queue.clone(), + f.scans(RecoveryScanLimits { + page: 1, + interval: Duration::from_secs(1), + }), + f.authority(), + maintenance(&f.handle, f.repository).await?, + )?; + let observer = timeout(Duration::from_secs(10), async { + loop { + if let Some(observer) = queue.pending(saved.token().operation).await { + break observer; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert!(matches!( + timeout(Duration::from_secs(10), observer.wait()).await?, + PublicationState::Uncertain(_) + )); + let scan = scanner.shutdown().await?; + assert_eq!(scan.release_submitted, 1); + assert_eq!(queue.stats().await.admitted, 1); + f.handle + .query(0, 128, |db| { + assert_eq!( + db.query_row("SELECT count(*) FROM catalog_leases", [], |r| r + .get::<_, u64>(0))?, + 0 + ); + Ok(Vec::new()) + }) + .await?; + assert_eq!(queue.close_and_drain().await.len(), 1); + let scanner = RecoverySupervisor::start_retiring( + f.client(), + f.target.clone(), + (*store).clone(), + queue.clone(), + f.scans(RecoveryScanLimits { + page: 1, + interval: Duration::from_secs(1), + }), + f.authority(), + maintenance(&f.handle, f.repository).await?, + )?; + timeout(Duration::from_secs(10), async { + while scanner.stats().release_recovered == 0 { + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await?; + assert!( + matches!(timeout(Duration::from_secs(10), observer.wait()).await?, PublicationState::Finished(Ok(PublicationOutcome::TerminalRelease(ref value))) if value.output==TerminalReleaseReply::Released) + ); + let scan = scanner.shutdown().await?; + assert_eq!(scan.release_submitted, 0); + assert_eq!(queue.stats().await.admitted, 0); + assert!(queue.close_and_drain().await.is_empty()); + drop(prepared); + cleaned(root.path(), &budget).await?; + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/inputs.rs b/crates/canopy-server/src/packs/publication/tests/inputs.rs index c35dad57..4d6a89a2 100644 --- a/crates/canopy-server/src/packs/publication/tests/inputs.rs +++ b/crates/canopy-server/src/packs/publication/tests/inputs.rs @@ -48,7 +48,11 @@ pub(super) async fn active( fixture: &Fixture, operation: [u8; 16], ) -> Result<(StagingCoordinator, StagingTicket)> { - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::new( fixture.client(), fixture.target.clone(), @@ -408,7 +412,11 @@ async fn source_pin_expiry_after_reconstruction_refuses_final_adoption() -> Resu .await?; old_ticket.stop(); assert!(old_coordinator.close_and_drain().await.is_empty()); - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::claim( fixture.client(), fixture.target.clone(), @@ -513,6 +521,7 @@ async fn bound_preparation_claim_adopts_exact_input_root_without_copying_nodes() actor: "owner".into(), }, Some(claimed.receipt), + fixture.authority(), ) .await?, ); @@ -523,8 +532,11 @@ async fn bound_preparation_claim_adopts_exact_input_root_without_copying_nodes() adopted.token()?.artifact_operation, prior.token()?.artifact_operation ); - let publisher = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let publisher = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let ready = session.ready_inputs(identity()?, adopted.clone()).await?; let registered = publisher.submit(ready).await?; let PublicationState::Finished(Ok(PublicationOutcome::Inputs(result))) = @@ -552,8 +564,11 @@ async fn claimed_staging_retains_exact_dispatch_after_absence_lost_ack_and_panic }; old_ticket.stop(); assert!(old_coordinator.close_and_drain().await.is_empty()); - let coordinator = - StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; coordinator.fault_for_test(fault); let ready = ReadyStaging::claim( fixture.client(), @@ -587,7 +602,10 @@ async fn claimed_staging_retains_exact_dispatch_after_absence_lost_ack_and_panic let retained = coordinator .pending(old.token.operation) .ok_or("retained claim")?; - assert_eq!(coordinator.stats().command_bytes, 8192); + assert_eq!( + coordinator.stats().command_bytes, + super::super::custody::RESERVATION + ); coordinator.recover(&retained)?; let StagingState::Active(next) = timeout(Duration::from_secs(10), retained.wait()).await? else { @@ -672,7 +690,11 @@ async fn restored_owner_claims_and_adopts_only_a_retained_exact_input_checkpoint check(&client, &fixture.target, proof.token()?).await?, Some(proof.clone()) ); - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::claim( client.clone(), fixture.target.clone(), @@ -859,7 +881,10 @@ async fn service_checkpoint_retains_exact_identity_after_cancellation_absence_lo assert!( matches!(retained.wait().await, Err(e) if matches!(&*e, StagingError::Checkpoint(_))) ); - assert_eq!(coordinator.stats().command_bytes, 3 * 4096); + assert_eq!( + coordinator.stats().command_bytes, + super::super::custody::RESERVATION + 4096 + ); // Closing retains unknown registrations and their charged command. let pending = timeout(Duration::from_secs(10), coordinator.close_and_drain()).await?; assert_eq!(pending.len(), 1); diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/bound.rs b/crates/canopy-server/src/packs/publication/tests/inputs/bound.rs index 88a4f451..071fc090 100644 --- a/crates/canopy-server/src/packs/publication/tests/inputs/bound.rs +++ b/crates/canopy-server/src/packs/publication/tests/inputs/bound.rs @@ -44,8 +44,11 @@ impl Bound { return Err("bound source".into()); }; assert!(staging.close_and_drain().await.is_empty()); - let coordinator = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let claimed = coordinator .submit( ReadyPreparation::claim( @@ -59,6 +62,7 @@ impl Bound { lease_ms: DEFAULT_LEASE_MS, }, identity()?, + fixture.authority(), ) .await?, ) @@ -193,7 +197,11 @@ async fn bound_checkpoint_canceled_observer_and_foreign_duplicate_closed_admissi let mutation = identity()?; let ready = session.ready_inputs(mutation, proof.clone()).await?; let other = Fixture::new(ObjectFormat::Sha256).await?; - let foreign = PublicationCoordinator::new(other.target.clone(), PublicationLimits::default())?; + let foreign = PublicationCoordinator::new( + other.target.clone(), + PublicationLimits::default(), + other.publication_budget.clone(), + )?; let refused = foreign .submit(ready) .await @@ -247,6 +255,7 @@ async fn bound_checkpoint_canceled_observer_and_foreign_duplicate_closed_admissi actor: "owner".into(), }, Some(result.registration.receipt), + fixture.authority(), ) .await?, ); @@ -423,6 +432,7 @@ async fn bound_checkpoint_real_retained_pair_publishes_after_source_pin_expiry_i indexes, files, Some(registered.registration.receipt), + bound.fixture.authority(), ) .await?, ); @@ -472,6 +482,7 @@ async fn bound_checkpoint_shares_push_actor_quotas_and_exact_mixed_byte_admissio maintenance_in_flight: 1, foreground_burst: 3, }, + bound.fixture.publication_budget.clone(), )?; mutate(&bound.fixture.handle, "INSERT INTO repository_members VALUES('writer','write'); INSERT INTO repository_members VALUES('third','write')".into()).await?; let (release, entered) = bound.coordinator.pause_for_test().await; diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/custody.rs b/crates/canopy-server/src/packs/publication/tests/inputs/custody.rs index 23f57b1c..637a743e 100644 --- a/crates/canopy-server/src/packs/publication/tests/inputs/custody.rs +++ b/crates/canopy-server/src/packs/publication/tests/inputs/custody.rs @@ -62,8 +62,11 @@ impl Recovered { checkpoint.wait().await.map_err(|e| e.to_string())?; old_ticket.stop(); assert!(old_coordinator.close_and_drain().await.is_empty()); - let coordinator = - StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::claim( fixture.client(), fixture.target.clone(), diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/requests.rs b/crates/canopy-server/src/packs/publication/tests/inputs/requests.rs index ae37e7d4..27608b7a 100644 --- a/crates/canopy-server/src/packs/publication/tests/inputs/requests.rs +++ b/crates/canopy-server/src/packs/publication/tests/inputs/requests.rs @@ -116,8 +116,11 @@ impl Request { operation, ) .await?; - let coordinator = - StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ticket = coordinator .submit( ReadyStaging::new( @@ -371,8 +374,11 @@ async fn request_checkpoint_owner_restore_adopts_original_bytes_after_source_pin ) .await?; let client = CellClient::local(request.fixture.registry.clone(), handle.clone()); - let coordinator = - StagingCoordinator::new(request.fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + request.fixture.target.clone(), + StagingLimits::default(), + request.fixture.authority(), + )?; let old = retained.token()?; let ticket = coordinator .submit( @@ -583,7 +589,10 @@ async fn request_checkpoint_append_recovers_exact_uncertain_registration_and_old .register_inputs(next.clone(), identity()?) .is_err() ); - assert_eq!(request.coordinator.stats().command_bytes, 12 << 10); + assert_eq!( + request.coordinator.stats().command_bytes, + super::super::super::custody::RESERVATION + 4096 + ); request.coordinator.recover(&request.ticket)?; let registered = request .ticket diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/requests/results.rs b/crates/canopy-server/src/packs/publication/tests/inputs/requests/results.rs index a4fbf2cd..e572d0a1 100644 --- a/crates/canopy-server/src/packs/publication/tests/inputs/requests/results.rs +++ b/crates/canopy-server/src/packs/publication/tests/inputs/requests/results.rs @@ -324,21 +324,21 @@ async fn root_outcome_preserves_plain_http_errors_without_verifying_or_publishin .await? .finish() .await?; - request - .fixture - .client() - .command::( - &request.fixture.target, - identity()?, - initial.empty_ref_initialization().await?, - ) - .await?; + let (command, _) = super::super::super::initialization::registered( + &request.fixture, + &initial, + initial.empty_ref_initialization().await?, + identity()?, + ) + .await?; + command.execute().await?; drop(initial); super::super::super::prepare::cleaned(initial_root.path(), &initial_budget).await?; let before = super::super::super::publishing::state(&request.fixture.handle).await?; let p = PublicationCoordinator::new( request.fixture.target.clone(), PublicationLimits::default(), + request.fixture.publication_budget.clone(), )?; let observer = request.ticket.publish(&p, ready)?; let PublicationState::Finished(Ok(PublicationOutcome::RootPush(committed))) = @@ -414,6 +414,7 @@ async fn root_outcome_exact_recovery_preserves_commits_and_refuses_expired_input let p = PublicationCoordinator::new( request.fixture.target.clone(), PublicationLimits::default(), + request.fixture.publication_budget.clone(), )?; p.fault_for_test(fault); drop(request.ticket.publish(&p, ready)?); diff --git a/crates/canopy-server/src/packs/publication/tests/mandatory_registration.rs b/crates/canopy-server/src/packs/publication/tests/mandatory_registration.rs index afc0f659..383512d9 100644 --- a/crates/canopy-server/src/packs/publication/tests/mandatory_registration.rs +++ b/crates/canopy-server/src/packs/publication/tests/mandatory_registration.rs @@ -140,8 +140,9 @@ pub(super) async fn qualify(context: Context<'_>) -> Result { not_started(f, refusal_command).await?; let original = Box::pin(command.clone().execute()).await?; assert!(matches!(original.output, RefPolicyReply::Registered(value) if value.valid)); - let PublicationOutcome::PolicyPage(recovered) = - registered.dispatch_any(&f.client(), store, &flag).await? + let PublicationOutcome::PolicyPage(recovered) = registered + .dispatch_any(&f.client(), store, &f.authority(), &flag) + .await? else { return Err("registered page lost its original result".into()); }; @@ -178,8 +179,9 @@ pub(super) async fn qualify(context: Context<'_>) -> Result { matches!(&original.output, RootCompletionReply::Completed(value) if !value.completion.rejected && value.completion.publication.is_some()) ); - let PublicationOutcome::RootPush(recovered) = - registered.dispatch_any(&f.client(), store, &flag).await? + let PublicationOutcome::RootPush(recovered) = registered + .dispatch_any(&f.client(), store, &f.authority(), &flag) + .await? else { return Err("registered root lost its original result".into()); }; diff --git a/crates/canopy-server/src/packs/publication/tests/native_capture.rs b/crates/canopy-server/src/packs/publication/tests/native_capture.rs index 53964261..c0fff050 100644 --- a/crates/canopy-server/src/packs/publication/tests/native_capture.rs +++ b/crates/canopy-server/src/packs/publication/tests/native_capture.rs @@ -456,13 +456,24 @@ async fn native_receive_case(format: ObjectFormat, rooted: bool, mode: Completio .await? .finish() .await?; - fixture - .client() - .command::( - &fixture.target, + let (command, registered) = super::initialization::registered( + &fixture, + &empty, + empty.empty_ref_initialization().await?, + identity()?, + ) + .await?; + command.execute().await?; + // Match actual startup before building later native policy work. + registered + .ready_terminal_release( + fixture.client(), + &store, + super::terminal_retention::maintenance(&fixture.handle, fixture.repository).await?, identity()?, - empty.empty_ref_initialization().await?, ) + .await? + .complete() .await?; drop(empty); cleaned(root.path(), &budget).await?; @@ -565,17 +576,11 @@ async fn native_receive_case(format: ObjectFormat, rooted: bool, mode: Completio ) .await?; let request_digest = encoded.identity().request_digest; - let mut staging_limits = StagingLimits::default(); - if matches!( - mode, - CompletionMode::Dispatch { - loss: super::root_dispatch::Loss::Expiry, - .. - } - ) { - staging_limits.bound_lifetime_ms = 5000; - } - let coordinator = StagingCoordinator::new(fixture.target.clone(), staging_limits)?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::new( fixture.client(), fixture.target.clone(), @@ -623,7 +628,7 @@ async fn native_receive_case(format: ObjectFormat, rooted: bool, mode: Completio .await .map_err(|error| StagingError::Input(Box::new(error)))?; let response = producer - .run_native_receive(preflight.into_native_request()) + .run_native_receive(&context, preflight.into_native_request()) .await .map_err(|error| StagingError::Input(Box::new(error)))?; let inputs = producer @@ -832,17 +837,30 @@ async fn native_receive_case(format: ObjectFormat, rooted: bool, mode: Completio } let physical_root = Arc::new(tempfile::TempDir::new()?); let physical_disk = DiskBudget::new(256 << 20); - let mut verifier = PhysicalVerifier::download( - physical_root.path(), - physical_disk.clone(), - &store, - inputs[0], - physical_limits(), - native.scope(NativeClass::Foreground), - ) - .await?; - let segment = verifier.inspect_next_shard(inputs[0].object_count).await?; - let witness = verifier.finish().await?; + let verify_root = physical_root.clone(); + let verify_disk = physical_disk.clone(); + let verify_store = store.clone(); + let verify_native = native.clone(); + let work = ticket.spawn(move |context| async move { + let result = async { + let mut verifier = PhysicalVerifier::download_staged( + &context, + verify_root.path(), + verify_disk, + &verify_store, + inputs[0], + physical_limits(), + verify_native.scope(NativeClass::Foreground), + ) + .await?; + let segment = verifier.inspect_next_shard(inputs[0].object_count).await?; + let witness = verifier.finish().await?; + Ok::<_, crate::packs::verification::PhysicalError>((witness, segment)) + } + .await; + result.map_err(|error| StagingError::Input(Box::new(error))) + })?; + let (witness, segment) = work.wait().await.map_err(|error| error.to_string())?; ticket.seal()?; assert!(matches!( timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, @@ -930,7 +948,7 @@ async fn native_receive_case(format: ObjectFormat, rooted: bool, mode: Completio let producer_root = Arc::clone(&physical_root); let producer_disk = physical_disk.clone(); let publication_identity = identity()?; - let work = ticket.spawn_bound(move |_| async move { + let work = ticket.spawn_bound(move |_, _context| async move { let result = async { let mut builder = CatalogPreparation::new( producer_root.path(), @@ -960,8 +978,11 @@ async fn native_receive_case(format: ObjectFormat, rooted: bool, mode: Completio .wait() .await .map_err(|error| format!("native receive stage: {error:?}"))?; - let publications = - PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let publications = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; let observer = ticket.publish(&publications, ready)?; let completed = super::coordinator::finished(timeout(Duration::from_secs(10), observer.wait()).await?)?; @@ -1093,7 +1114,11 @@ async fn native_capture_rejects_scope_limits_mutation_and_active_native_workers( Arc::new(InMemory::new()), fixture.repository, )); - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::new( fixture.client(), fixture.target.clone(), diff --git a/crates/canopy-server/src/packs/publication/tests/policy_dispatch.rs b/crates/canopy-server/src/packs/publication/tests/policy_dispatch.rs index f8940f57..5c33075b 100644 --- a/crates/canopy-server/src/packs/publication/tests/policy_dispatch.rs +++ b/crates/canopy-server/src/packs/publication/tests/policy_dispatch.rs @@ -66,7 +66,11 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu ); let refusal_evidence = refusal.evidence_for_test(); let operation = prepared.token().operation; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let before = state(&f.handle).await?; let mut head = None; let mut offset = 0; @@ -97,7 +101,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu let (release, wait) = tokio::sync::oneshot::channel(); let (entered, running) = tokio::sync::oneshot::channel(); let worker = if offset == 0 { - Some(ticket.spawn_bound(move |_| async move { + Some(ticket.spawn_bound(move |_, _context| async move { let _ = entered.send(()); wait.await.map_err(|_| StagingError::Worker)?; Ok(42u64) @@ -329,7 +333,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu assert_eq!(body, rejected.body); assert_eq!(body.windows(3).filter(|part| *part == b"ng ").count(), 257); assert!(!body.windows(3).any(|part| part == b"ok ")); - assert_eq!(f.counts().await?, (0, 2)); + assert_eq!(f.counts_for(prepared.token()).await?, (0, 1)); roots_unchanged(&before, &state(&f.handle).await?)?; } PublicationState::Finished(Err(error)) => { @@ -344,7 +348,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu )); assert_denied_page(f, &evidence, loss).await?; assert!(observer.root_response(store).await.is_err()); - assert_eq!(f.counts().await?, (1, 2)); + assert_eq!(f.counts_for(prepared.token()).await?, (1, 1)); roots_unchanged(&before, &state(&f.handle).await?)?; } other => return Err(format!("unexpected policy result {other:?}").into()), @@ -476,12 +480,17 @@ async fn complete( body.extend_from_slice(&part); } assert_eq!(body, expected.body); - // Initialization and the native attempt each retain an independent pin. - assert_eq!(f.counts().await?, (0, 2)); + // Initialization is retired; the native attempt retains its own pin. + assert_eq!(f.counts_for(prepared.token()).await?, (0, 1)); + let operation = prepared.token().operation; f.handle - .query(0, 1024, |db| { + .query(0, 1024, move |db| { assert_eq!( - db.query_row("SELECT count(*) FROM pushes", [], |r| r.get::<_, u64>(0))?, + db.query_row( + "SELECT count(*) FROM pushes WHERE id=?1 AND response_root IS NOT NULL", + [operation.as_slice()], + |r| r.get::<_, u64>(0) + )?, 1 ); assert_eq!( diff --git a/crates/canopy-server/src/packs/publication/tests/policy_refusal.rs b/crates/canopy-server/src/packs/publication/tests/policy_refusal.rs index 315fd4ce..0952b083 100644 --- a/crates/canopy-server/src/packs/publication/tests/policy_refusal.rs +++ b/crates/canopy-server/src/packs/publication/tests/policy_refusal.rs @@ -45,6 +45,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu let expected = crate::push::report::rejected_report(&request.response, crate::push::report::REJECTED)?; let session = Arc::new(ticket.bound_session()?); + let attempt_token = session.check.token; let operation = session.check.token.operation; assert!( session @@ -96,7 +97,11 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu page.refusal_fault_for_test(fault); let registered = page.persist_recovery(store, identity()?, None).await?; let page = page.bind_recovery(registered, store)?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; p.fault_for_test(1); drop(ticket.register_policy_page(&p, page)?); let StagingState::Uncertain(error) = settled(ticket, true).await? else { @@ -218,7 +223,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu assert!(matches!(outcome, PublicationState::Finished(Err(error)) if matches!(&*error, PublicationError::RootPush(InvocationError::Rejected(value)) if value.output==RootCompletionReply::Denied(PreparationDenial::Expired)))); assert!(observer.root_response(store).await.is_err()); - assert_eq!(f.counts().await?, (1, 2)); + assert_eq!(f.counts_for(attempt_token).await?, (1, 1)); } else { let PublicationState::Finished(Ok(PublicationOutcome::RootPush(committed))) = outcome else { @@ -256,7 +261,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu assert_eq!(body, expected.body); assert_eq!(body.windows(3).filter(|p| *p == b"ng ").count(), 257); assert!(!body.windows(3).any(|p| p == b"ok ")); - assert_eq!(f.counts().await?, (0, 2)); + assert_eq!(f.counts_for(attempt_token).await?, (0, 1)); } let Resolution::Committed(page) = f.client().resolve(&page_evidence).await? else { return Err("original registered page receipt missing".into()); @@ -340,13 +345,18 @@ async fn qualify_live(context: Context<'_>, loss: Loss) -> Result { .await?, ); let session = Arc::new(ticket.bound_session()?); + let attempt_token = session.check.token; let refusal = Arc::new( session .ready_root_refusal(identity()?, store, root, budget.clone(), None) .await?, ); let evidence = refusal.evidence_for_test(); - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let mut head = None; for offset in [0, 128, 256] { let page = pending @@ -471,7 +481,7 @@ async fn qualify_live(context: Context<'_>, loss: Loss) -> Result { let after: Vec = serde_json::from_slice(&state(&f.handle).await?)?; assert_eq!(before[..6], after[..6]); } - assert_eq!(f.counts().await?, (0, 2)); + assert_eq!(f.counts_for(attempt_token).await?, (0, 1)); assert!(staging.close_and_drain().await.is_empty()); assert!(p.close_and_drain().await.is_empty()); assert_eq!(p.reservations_for_test().await, (0, 0, 0)); @@ -492,7 +502,7 @@ async fn owned_positive( let directory = root.to_path_buf(); let mutation = identity()?; Ok(ticket - .spawn_bound(move |_| async move { + .spawn_bound(move |_, _context| async move { owner .ready_root_push(mutation, &guard, &directory, budget, limits(), None) .await diff --git a/crates/canopy-server/src/packs/publication/tests/preparation_receipt.rs b/crates/canopy-server/src/packs/publication/tests/preparation_receipt.rs new file mode 100644 index 00000000..3de5b8d8 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/preparation_receipt.rs @@ -0,0 +1,543 @@ +//! First preparation admission survives transport loss independently of custody. +use super::publishing::edit; +use super::*; +use cellule_runtime::Resolution; +use tokio::time::{Duration, timeout}; + +fn denied( + result: std::result::Result< + cellule_runtime::Committed, + InvocationError, + >, + reason: PreparationDenial, +) { + match result { + Err(InvocationError::Rejected(value)) => { + assert_eq!(value.output, PreparationReply::Denied(reason)) + } + other => panic!("expected preparation denial {reason:?}, observed {other:?}"), + } +} + +async fn expire(expires_at_ms: i64) -> Result { + let now = sql::now(0)?; + if now <= expires_at_ms { + tokio::time::sleep(Duration::from_millis(u64::try_from( + expires_at_ms - now + 1, + )?)) + .await; + } + Ok(()) +} + +#[tokio::test] +async fn initial_preparation_receipt_is_saved_with_the_actual_admission() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let original = f + .client() + .command::(&f.target, identity()?, f.begin([218; 16])) + .await?; + let accepted = lease(original.output)?; + let operation = accepted.token.operation; + let saved = f + .handle + .query(0, 2048, move |db| { + Ok(db.query_row( + "SELECT initial_preparation FROM pushes WHERE id=?1", + [operation.as_slice()], + |row| row.get::<_, Vec>(0), + )?) + }) + .await?; + assert!(!saved.is_empty()); + assert!(saved.len() <= CERTIFICATE_BYTES as usize); + assert_eq!(accepted.token.attempt, original.receipt.commit_sequence); + assert_eq!(f.counts().await?, (1, 1)); + let known = PreparationAdmission::load(&f.client(), &f.target, operation) + .await? + .ok_or("initial receipt absent")?; + assert_eq!(known.lease(), accepted); + assert_eq!(known.receipt(), original.receipt); + assert_eq!(known.request(), &f.begin(operation)); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn initial_preparation_receipt_late_write_rolls_back_and_first_result_is_immutable() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for bound in [false, true] { + let f = Fixture::new(format).await?; + let input = f.begin([219; 16]); + let prior = if bound { + let staged = f + .client() + .command::(&f.target, identity()?, input.clone()) + .await?; + let StagingReply::Granted(staged) = staged.output else { + return Err("staging grant absent".into()); + }; + f.client() + .command::(&f.target, identity()?, check(staged.token)) + .await?; + Some(staged.token) + } else { + None + }; + let command = f + .client() + .prepare_command::(&f.target, identity()?, input.clone()) + .await?; + let evidence = command.evidence().clone(); + let event = if bound { + "UPDATE OF initial_preparation" + } else { + "INSERT" + }; + edit(&f, &format!("CREATE TRIGGER initial_preparation_fault BEFORE {event} ON pushes BEGIN SELECT RAISE(ABORT,'late initial preparation receipt failure'); END")).await?; + assert!(command.clone().execute().await.is_err()); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + assert_eq!(f.counts().await?, if bound { (1, 1) } else { (0, 0) }); + assert!( + PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .is_none() + ); + edit(&f, "DROP TRIGGER initial_preparation_fault").await?; + let original = command.execute().await?; + let accepted = lease(original.output.clone())?; + assert_eq!(accepted.token.artifact_operation, artifact_number(1)); + if let Some(prior) = prior { + assert_eq!(accepted.token, prior); + assert!(original.receipt.commit_sequence > prior.attempt); + assert_eq!( + StagingAdmission::load(&f.client(), &f.target, input.operation) + .await? + .ok_or("staging knowledge lost")? + .lease() + .token, + prior + ); + } + let saved = PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .ok_or("initial receipt absent")?; + assert_eq!( + saved.original(&evidence)?.ok_or("original stamp absent")?, + original + ); + let observer = f + .client() + .prepare_command::(&f.target, identity()?, input.clone()) + .await?; + assert!(saved.original(observer.evidence())?.is_none()); + let later = observer.execute().await?; + assert_eq!(lease(later.output)?.token, accepted.token); + assert!(later.receipt.commit_sequence > original.receipt.commit_sequence); + assert_eq!( + PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .ok_or("original replaced")? + .receipt(), + original.receipt + ); + for statement in [ + "UPDATE pushes SET initial_preparation=NULL", + "UPDATE pushes SET initial_preparation=x'01'", + "DELETE FROM pushes", + "INSERT OR REPLACE INTO pushes(id,actor,request_digest) SELECT id,actor,request_digest FROM pushes", + ] { + assert!(edit(&f, statement).await.is_err()); + } + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn initial_preparation_receipt_cold_owner_and_sdk_expiry_preserve_actual_receipt() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let input = f.begin([220; 16]); + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1000; + let command = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginPreparation(input.clone()), + mutation, + ) + .await?; + let evidence = command.evidence().clone(); + // Discard every factory after acceptance; cold Claim starts from the + // mandatory registered predecessor, never a raw command adapter. + let original = command + .register(&f.client(), identity()?) + .await? + .recover_preparation(&f.client()) + .await?; + let old = lease(original.output.clone())?; + let (runtime, handle, client) = + super::durable_recovery::restore_owner(&f, &check(old.token)).await?; + expire(mutation.expires_at_ms).await?; + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Expired + )); + let known = PreparationAdmission::load(&client, &f.target, input.operation) + .await? + .ok_or("cold receipt absent")?; + assert_eq!( + known.original(&evidence)?.ok_or("cold original absent")?, + original + ); + assert_eq!(known.lease(), old); + denied( + client + .command::(&f.target, identity()?, request(old.token)) + .await, + PreparationDenial::Stale, + ); + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let ticket = coordinator + .submit( + known + .ready_claim(client.clone(), DEFAULT_LEASE_MS, identity()?, f.authority()) + .await?, + ) + .await?; + let PublicationState::Finished(Ok(PublicationOutcome::Preparation(next))) = + timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("cold Claim did not finish".into()); + }; + let next = next.session.map_err(|error| error.to_string())?; + assert_eq!(next.lease.token.owner, handle.owner_fence()); + assert_ne!(next.lease.token.owner, old.token.owner); + assert_ne!( + next.lease.token.artifact_operation, + old.token.artifact_operation + ); + assert_eq!(counts(&handle).await?, (1, 2)); + let preserved = PreparationAdmission::load(&client, &f.target, input.operation) + .await? + .ok_or("Claim lost initial receipt")?; + assert_eq!( + preserved + .original(&evidence)? + .ok_or("initial identity lost")?, + original + ); + assert!(coordinator.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn initial_preparation_receipt_reaped_restart_claim_checks_original_and_rolls_back() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let input = f.begin([221; 16]); + let original = f + .client() + .command::(&f.target, identity()?, input.clone()) + .await?; + let old = lease(original.output.clone())?; + edit(&f, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0").await?; + assert_eq!( + f.client() + .command::( + &f.target, + identity()?, + MaintenanceRequest { + repository: f.repository, + actor: "owner".into(), + owner: f.handle.owner_fence() + } + ) + .await? + .output, + 2 + ); + assert_eq!(f.counts().await?, (0, 0)); + let known = PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .ok_or("reaping lost receipt")?; + assert_eq!(known.receipt(), original.receipt); + assert!( + PreparationSession::open( + f.client(), + f.target.clone(), + check(old.token), + Some(original.receipt), + f.authority(), + ) + .await + .is_err() + ); + for variant in 0..4 { + let mut forged = request(old.token); + match variant { + 0 => forged.check.token.owner.epoch += 1, + 1 => forged.check.token.attempt += 1, + 2 => forged.check.token.artifact_operation = artifact_number(71), + _ => forged.check.token.request_digest = [72; 32], + } + denied( + f.client() + .command::(&f.target, identity()?, forged) + .await, + PreparationDenial::Missing, + ); + assert_eq!(f.counts().await?, (0, 0)); + } + let command = f + .client() + .prepare_command::(&f.target, identity()?, request(old.token)) + .await?; + let evidence = command.evidence().clone(); + edit(&f, "CREATE TRIGGER preparation_restart_fault BEFORE INSERT ON catalog_operations BEGIN SELECT RAISE(ABORT,'late preparation restart failure'); END").await?; + assert!(command.clone().execute().await.is_err()); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + assert_eq!(f.counts().await?, (0, 0)); + edit(&f, "DROP TRIGGER preparation_restart_fault").await?; + let next = lease(command.execute().await?.output)?; + assert_eq!(next.token.artifact_operation, artifact_number(2)); + assert_eq!(next.token.owner, f.handle.owner_fence()); + assert_eq!(f.counts().await?, (1, 1)); + // The original grant cannot displace a known active successor. + denied( + f.client() + .command::(&f.target, identity()?, request(old.token)) + .await, + PreparationDenial::Stale, + ); + assert_eq!( + PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .ok_or("restart replaced receipt")? + .lease(), + old + ); + f.client() + .command::(&f.target, identity()?, check(next.token)) + .await?; + edit( + &f, + "UPDATE pushes SET response_id=zeroblob(16),completion_digest=zeroblob(32),rejected=1", + ) + .await?; + denied( + f.client() + .command::(&f.target, identity()?, request(old.token)) + .await, + PreparationDenial::Conflict, + ); + assert_eq!(f.counts().await?, (0, 1)); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn initial_preparation_receipt_knowledge_does_not_restore_revoked_or_expired_custody() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for revoked in [false, true] { + let f = Fixture::new(format).await?; + let input = f.begin([222; 16]); + let command = f + .client() + .prepare_command::(&f.target, identity()?, input.clone()) + .await?; + let evidence = command.evidence().clone(); + let original = command.execute().await?; + let old = lease(original.output.clone())?; + if revoked { + edit( + &f, + "UPDATE repository_identity SET owner='other' WHERE singleton=1", + ) + .await?; + } else { + edit(&f, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0").await?; + } + let known = PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .ok_or("custody hid knowledge")?; + assert_eq!( + known.original(&evidence)?.ok_or("original hidden")?, + original + ); + assert!( + PreparationSession::open( + f.client(), + f.target.clone(), + check(old.token), + Some(original.receipt), + f.authority(), + ) + .await + .is_err() + ); + denied( + f.client() + .command::(&f.target, identity()?, request(old.token)) + .await, + if revoked { + PreparationDenial::Unauthorized + } else { + PreparationDenial::Expired + }, + ); + if revoked { + denied( + f.client() + .command::(&f.target, identity()?, request(old.token)) + .await, + PreparationDenial::Unauthorized, + ); + } + assert_eq!(f.counts().await?, (1, 1)); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn initial_preparation_receipt_rejects_corruption_and_cross_purpose_metadata() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let input = f.begin([223; 16]); + let staged = f + .client() + .command::(&f.target, identity()?, input.clone()) + .await?; + let StagingReply::Granted(staged) = staged.output else { + return Err("staging grant absent".into()); + }; + f.client() + .command::(&f.target, identity()?, check(staged.token)) + .await?; + let original = f + .client() + .command::(&f.target, identity()?, input.clone()) + .await?; + let body = f + .handle + .query(0, 2048, |db| { + Ok( + db.query_row("SELECT initial_preparation FROM pushes", [], |row| { + row.get::<_, Vec>(0) + })?, + ) + }) + .await?; + let mut corrupt = body.clone(); + *corrupt.last_mut().ok_or("empty receipt")? ^= 1; + edit(&f, "DROP TRIGGER push_initial_preparation_immutable").await?; + edit( + &f, + &format!( + "UPDATE pushes SET initial_preparation=x'{}'", + hex::encode(corrupt) + ), + ) + .await?; + assert!( + PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await + .is_err() + ); + edit( + &f, + &format!( + "UPDATE pushes SET initial_preparation=x'{}'", + hex::encode(body) + ), + ) + .await?; + assert_eq!( + PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .ok_or("restored receipt absent")? + .receipt(), + original.receipt + ); + // Keep scope, actor, operation, digest and MAC valid. Only the purpose + // differs, so a context mismatch cannot satisfy this assertion. + edit(&f, "UPDATE pushes SET initial_preparation=initial_staging").await?; + assert!(matches!( + PreparationAdmission::load(&f.client(), &f.target, input.operation).await, + Err(PreparationReceiptError::Codec(CodecError::Invalid( + "initial admission receipt purpose" + ))) + )); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn initial_preparation_receipt_does_not_invent_knowledge_for_denied_or_unexecuted_begin() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let mut input = f.begin([225; 16]); + input.actor = "outsider".into(); + denied( + f.client() + .command::(&f.target, identity()?, input.clone()) + .await, + PreparationDenial::Unauthorized, + ); + assert!( + PreparationAdmission::load(&f.client(), &f.target, input.operation) + .await? + .is_none() + ); + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1000; + let command = f + .client() + .prepare_command::(&f.target, mutation, f.begin([226; 16])) + .await?; + let evidence = command.evidence().clone(); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + drop(command); + expire(mutation.expires_at_ms).await?; + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + // No domain record does not convert Expired into authoritative absence. + assert!( + PreparationAdmission::load(&f.client(), &f.target, [226; 16]) + .await? + .is_none() + ); + assert_eq!(f.counts().await?, (0, 0)); + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/prepare.rs b/crates/canopy-server/src/packs/publication/tests/prepare.rs index 97d16ca5..40dd82aa 100644 --- a/crates/canopy-server/src/packs/publication/tests/prepare.rs +++ b/crates/canopy-server/src/packs/publication/tests/prepare.rs @@ -50,10 +50,7 @@ pub(super) async fn opened( Arc, Arc, )> { - let started = fixture - .client() - .command::(&fixture.target, identity()?, fixture.begin(operation)) - .await?; + let started = registered_preparation(fixture, operation).await?; let token = lease(started.output)?.token; let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), fixture.format)); let files = Arc::new(CatalogFiles::new( @@ -71,11 +68,50 @@ pub(super) async fn opened( Arc::clone(&indexes), Arc::clone(&files), Some(started.receipt), + fixture.authority(), ) .await?, ); Ok((base, files, indexes)) } +/// Keep fixture construction under actual service-owned custody renewals. +/// A long native history must not consume the lease before the tested action. +pub(super) async fn renewing( + fixture: &Fixture, + base: &PreparationBaseResolver, + work: impl std::future::Future>, +) -> Result { + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits::default(), + fixture.publication_budget.clone(), + )?; + tokio::pin!(work); + let mut ticks = tokio::time::interval(Duration::from_secs(10)); + ticks.tick().await; + let result = loop { + tokio::select! { + result = &mut work => break result, + _ = ticks.tick() => { + let renewed = async { + let ready = base.ready_renew(identity()?, DEFAULT_LEASE_MS).await?; + let ticket = coordinator.submit(ready).await?; + match ticket.wait().await { + PublicationState::Finished(Ok(PublicationOutcome::Preparation(value))) => { + value.session.map_err(|error| error.to_string())?; + Ok(()) + } + other => Err(format!("fixture renewal: {other:?}").into()), + } + }.await; + if let Err(error) = renewed { break Err(error); } + } + } + }; + assert!(coordinator.close_and_drain().await.is_empty()); + result +} + pub(super) async fn physical( prepared: &Prepared, root: &Path, diff --git a/crates/canopy-server/src/packs/publication/tests/publishing.rs b/crates/canopy-server/src/packs/publication/tests/publishing.rs index 3f05e5a4..f508b4ba 100644 --- a/crates/canopy-server/src/packs/publication/tests/publishing.rs +++ b/crates/canopy-server/src/packs/publication/tests/publishing.rs @@ -57,30 +57,34 @@ pub(super) async fn assembled( .ok_or("blob")? .0 .oid; - let (tip, other) = if depth == 0 { - (initial, initial) - } else { - history(&mut native, initial, depth).await? - }; - let root = tempfile::TempDir::new()?; - let budget = DiskBudget::new(256 << 20); - let mut builder = CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; - let (witness, segments) = physical(&native, root.path(), budget.clone()).await?; - builder.begin_pack(witness)?; - for segment in segments { - builder.add_segment(segment).await?; - } - builder.finish_pack().await?; - Ok(Graph { - prepared: builder.finish().await?, - root, - budget, - initial, - tip, - other, - blob, - store: native.store, + super::prepare::renewing(fixture, &base, async { + let (tip, other) = if depth == 0 { + (initial, initial) + } else { + history(&mut native, initial, depth).await? + }; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let mut builder = + CatalogPreparation::new(root.path(), budget.clone(), base.clone(), limits()).await?; + let (witness, segments) = physical(&native, root.path(), budget.clone()).await?; + builder.begin_pack(witness)?; + for segment in segments { + builder.add_segment(segment).await?; + } + builder.finish_pack().await?; + Ok(Graph { + prepared: builder.finish().await?, + root, + budget, + initial, + tip, + other, + blob, + store: native.store, + }) }) + .await } async fn history( native: &mut Prepared, @@ -206,10 +210,10 @@ pub(super) async fn state(handle: &CellHandle) -> Result> { } let policy_hash=*hash.finalize().as_bytes(); hash.update(b"\0completed-root-and-operation-state\0"); - let mut outcomes=connection.prepare("SELECT id,actor,request_digest,response_id,completion_digest,rejected,publication,publication_plan_digest,response_root FROM pushes ORDER BY id")?; + let mut outcomes=connection.prepare("SELECT id,actor,request_digest,response_id,completion_digest,rejected,publication,publication_plan_digest,response_root,initial_staging,initial_preparation FROM pushes ORDER BY id")?; let mut rows=outcomes.query([])?; while let Some(row)=rows.next()? { - let record=(row.get::<_,Vec>(0)?,row.get::<_,String>(1)?,row.get::<_,Vec>(2)?,row.get::<_,Option>>(3)?,row.get::<_,Option>>(4)?,row.get::<_,Option>(5)?,row.get::<_,Option>>(6)?,row.get::<_,Option>>(7)?,row.get::<_,Option>>(8)?); + let record=(row.get::<_,Vec>(0)?,row.get::<_,String>(1)?,row.get::<_,Vec>(2)?,row.get::<_,Option>>(3)?,row.get::<_,Option>>(4)?,row.get::<_,Option>(5)?,row.get::<_,Option>>(6)?,row.get::<_,Option>>(7)?,row.get::<_,Option>>(8)?,row.get::<_,Option>>(9)?,row.get::<_,Option>>(10)?); hash.update(&serde_json::to_vec(&record).map_err(|_|Error::Command("fixture root outcome hash"))?); } let mut operations=connection.prepare("SELECT id,actor,request_digest,artifact_operation,generation,attestation,attestation_digest FROM catalog_operations ORDER BY id")?; @@ -738,19 +742,25 @@ async fn later_policy_changes_and_late_transaction_failures_publish_nothing() -> .await?; edit(&fixture,"UPDATE branch_rules SET version=2,require_pull_request=1 WHERE reference='refs/heads/main';").await?; reject(&fixture, valid.clone(), PreparationDenial::Conflict).await?; - edit(&fixture,"UPDATE branch_rules SET version=3,require_pull_request=0 WHERE reference='refs/heads/main'; CREATE TRIGGER forced_publication_failure BEFORE INSERT ON pushes BEGIN SELECT RAISE(ABORT,'forced late publication failure'); END;").await?; + edit(&fixture,"UPDATE branch_rules SET version=3,require_pull_request=0 WHERE reference='refs/heads/main'; CREATE TRIGGER forced_publication_failure BEFORE UPDATE OF publication ON pushes BEGIN SELECT RAISE(ABORT,'forced late publication failure'); END;").await?; let before = state(&fixture.handle).await?; - let failed = fixture + let command = fixture .client() - .command::(&fixture.target, identity()?, valid.clone()) - .await; - assert!(failed.is_err(), "{failed:?}"); + .prepare_command::(&fixture.target, identity()?, valid) + .await?; + let evidence = command.evidence().clone(); + let failed = command.clone().execute().await; + assert!( + matches!(&failed, Err(InvocationError::NotStarted(error)) if format!("{error:?}").contains("forced late publication failure")), + "{failed:?}" + ); + assert!(matches!( + fixture.client().resolve(&evidence).await?, + cellule_runtime::Resolution::Absent + )); assert_eq!(state(&fixture.handle).await?, before); edit(&fixture, "DROP TRIGGER forced_publication_failure;").await?; - fixture - .client() - .command::(&fixture.target, identity()?, valid) - .await?; + command.execute().await?; drop(graph.prepared); cleaned(graph.root.path(), &graph.budget).await?; drop(next.prepared); @@ -933,7 +943,7 @@ async fn expired_and_claimed_proofs_and_mutable_publication_facts_fail_closed() "UPDATE pushes SET publication=zeroblob(53)", "UPDATE pushes SET publication_plan_digest=zeroblob(32)", "UPDATE pushes SET actor='outsider'", - "INSERT OR REPLACE INTO pushes SELECT id,actor,request_digest,options,response_id,rejected,rejection_reason,publication,publication_plan_digest FROM pushes", + "INSERT OR REPLACE INTO pushes SELECT * FROM pushes", "INSERT OR REPLACE INTO catalog_generations SELECT * FROM catalog_generations WHERE generation=1", ] { assert!(edit(&fixture, sql).await.is_err(), "{sql}"); diff --git a/crates/canopy-server/src/packs/publication/tests/recovery_discovery.rs b/crates/canopy-server/src/packs/publication/tests/recovery_discovery.rs index 49ae5bc7..dd246cb5 100644 --- a/crates/canopy-server/src/packs/publication/tests/recovery_discovery.rs +++ b/crates/canopy-server/src/packs/publication/tests/recovery_discovery.rs @@ -40,14 +40,19 @@ async fn restart_scan_seeks_bounded_keys_and_revisits_corrupt_pins_without_starv assert!(details.iter().all(|value| !value.contains("TEMP B-TREE")), "{details:?}"); Ok(Vec::new()) }).await?; - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let store = ArtifactStore::new(Arc::new(InMemory::new()), f.repository); let service = RecoverySupervisor::start( f.client(), f.target.clone(), store.clone(), queue.clone(), - scan_limits(17), + f.scans(scan_limits(17)), + f.authority(), )?; let stats = scanned(&service, |stats| stats.passes >= 2).await?; assert!(stats.scanned >= 600); @@ -76,7 +81,8 @@ async fn restart_scan_seeks_bounded_keys_and_revisits_corrupt_pins_without_starv f.target.clone(), foreign, queue.clone(), - scan_limits(1) + f.scans(scan_limits(1)), + f.authority(), ), Err(RootRecoveryError::Context) )); @@ -94,7 +100,8 @@ pub(super) async fn leaves_live_owner( f.target.clone(), store.clone(), queue.clone(), - scan_limits(1), + f.scans(scan_limits(1)), + f.authority(), super::terminal_retention::maintenance(&f.handle, f.repository).await?, )?; let stats = scanned(&service, |stats| stats.deferred > 0).await?; @@ -138,7 +145,11 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8) -> Result { // identity binding before bundle I/O; the later valid head still runs. edit(f, "INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,expires_at_ms,recovery) SELECT zeroblob(16),1,zeroblob(16),x'0000000000000001',randomblob(16),0,recovery FROM catalog_leases WHERE recovery IS NOT NULL LIMIT 1").await?; } - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; queue.fault_for_test(fault); let (release, entered) = queue.pause_for_test().await; let service = RecoverySupervisor::start( @@ -146,10 +157,11 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8) -> Result { f.target.clone(), store.clone(), queue.clone(), - RecoveryScanLimits { + f.scans(RecoveryScanLimits { page: 1, interval: Duration::from_secs(1), - }, + }), + f.authority(), )?; timeout(Duration::from_secs(10), entered).await??; let observer = queue @@ -178,10 +190,11 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8) -> Result { f.target.clone(), store.clone(), queue.clone(), - RecoveryScanLimits { + f.scans(RecoveryScanLimits { page: 1, interval: Duration::from_secs(1), - }, + }), + f.authority(), )? } else { service @@ -210,13 +223,18 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8) -> Result { // Destroy local SQL and factory state. A new owner's scanner recognizes // the settled head without dispatching or claiming an old-owner command. let (runtime, _, client) = super::durable_recovery::restore_owner(f, &check).await?; - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let service = RecoverySupervisor::start( client.clone(), f.target.clone(), store.clone(), queue.clone(), - scan_limits(1), + f.scans(scan_limits(1)), + f.authority(), )?; let stats = scanned(&service, |stats| stats.settled > 0).await?; assert_eq!(stats.submitted, 0); @@ -225,7 +243,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8) -> Result { .await? .ok_or("settled pin lost on owner restore")?; assert_eq!(restored.evidence(), &original); - let recovered = restored.dispatch(&client, store).await?; + let recovered = restored.dispatch(&client, store, &f.authority()).await?; assert_eq!(recovered.receipt, completed.receipt); assert_eq!(recovered.output, completed.output); let lookup = BeginRequest { @@ -256,10 +274,18 @@ pub(super) async fn advanced_head( original: &RegisteredRootRecovery, expected: &cellule_runtime::Committed, ) -> Result { - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; queue.fault_for_test(2); let observer = queue - .submit(original.clone().ready(client.clone(), store.clone())?) + .submit( + original + .clone() + .ready(client.clone(), store.clone(), f.authority())?, + ) .await .map_err(|error| format!("historical admission: {:?}", error.reason))?; assert!(matches!( @@ -272,7 +298,8 @@ pub(super) async fn advanced_head( f.target.clone(), store.clone(), queue.clone(), - scan_limits(1), + f.scans(scan_limits(1)), + f.authority(), )?; let stats = scanned(&service, |stats| stats.recovered > 0 || stats.deferred > 0).await?; assert!( @@ -312,3 +339,77 @@ pub(super) fn qualify_native<'a>( _ => unreachable!("native recovery qualifier role"), } } + +#[tokio::test] +async fn resident_scanners_pause_independently_and_resume_live_indexed_discovery() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + edit(&f, "WITH RECURSIVE n(x) AS (VALUES(1) UNION ALL SELECT x+1 FROM n WHERE x<30) INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,expires_at_ms,recovery) SELECT zeroblob(16),x,zeroblob(16),x'0000000000000001',randomblob(16),0,x'01' FROM n").await?; + edit(&f, "WITH RECURSIVE n(x) AS (VALUES(1) UNION ALL SELECT x+1 FROM n WHERE x<30) INSERT INTO catalog_custody_commands(purpose,operation,step,incarnation,request_id,intent) SELECT 0, CAST(printf('%016d',x) AS BLOB),0,zeroblob(16),CAST(printf('%016d',x) AS BLOB),x'01' FROM n").await?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let roots = RecoverySupervisor::start( + f.client(), + f.target.clone(), + ArtifactStore::new(Arc::new(InMemory::new()), f.repository), + queue.clone(), + f.scans(scan_limits(1)), + f.authority(), + )?; + let custody = CustodySupervisor::start( + f.client(), + f.target.clone(), + queue.clone(), + f.scans(scan_limits(1)), + f.authority(), + )?; + scanned(&roots, |stats| stats.scanned > 0).await?; + tokio::join!(roots.pause(), custody.pause()); + let root_before = roots.stats(); + let custody_before = custody.stats(); + assert_eq!(f.scan_budget.in_flight(), 0); + tokio::time::sleep(Duration::from_millis(50)).await; + assert_eq!(roots.stats().scanned, root_before.scanned); + assert_eq!(custody.stats().scanned, custody_before.scanned); + assert_eq!(roots.stats().passes, root_before.passes); + assert_eq!(custody.stats().passes, custody_before.passes); + assert_eq!(queue.stats().await.admitted, 0); + // Resuming one owner must not restart its independently paused sibling. + roots.resume(); + scanned(&roots, |stats| stats.scanned > root_before.scanned).await?; + assert_eq!(custody.stats().scanned, custody_before.scanned); + roots.pause().await; + let paused = roots.stats(); + custody.resume(); + timeout(Duration::from_secs(10), async { + while custody.stats().scanned <= custody_before.scanned { + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await?; + assert_eq!(roots.stats().scanned, paused.scanned); + custody.pause().await; + assert_eq!(f.scan_budget.in_flight(), 0); + roots.resume(); + custody.resume(); + scanned(&roots, |stats| stats.passes > root_before.passes).await?; + timeout(Duration::from_secs(10), async { + while custody.stats().passes <= custody_before.passes { + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await?; + let (root_final, custody_final) = tokio::join!(roots.shutdown(), custody.shutdown()); + let root_final = root_final?; + let custody_final = custody_final?; + assert_eq!(root_final.scanned, root_final.failures); + assert_eq!(custody_final.scanned, custody_final.failures); + assert_eq!(f.scan_budget.in_flight(), 0); + assert!(queue.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/ref_policy.rs b/crates/canopy-server/src/packs/publication/tests/ref_policy.rs index f85d9580..39d053d5 100644 --- a/crates/canopy-server/src/packs/publication/tests/ref_policy.rs +++ b/crates/canopy-server/src/packs/publication/tests/ref_policy.rs @@ -41,9 +41,20 @@ async fn rooted(format: ObjectFormat) -> Result<(Fixture, Graph)> { .finish() .await?; let proof = empty.empty_ref_initialization().await?; - fixture - .client() - .command::(&fixture.target, identity()?, proof) + let (command, registered) = + super::initialization::registered(&fixture, &empty, proof, identity()?).await?; + command.execute().await?; + // Match production startup: the completed initial fact keeps its metadata + // and original receipt, while its generation-zero pin is retired. + registered + .ready_terminal_release( + fixture.client(), + &store, + super::terminal_retention::maintenance(&fixture.handle, fixture.repository).await?, + identity()?, + ) + .await? + .complete() .await?; drop(empty); cleaned(root.path(), &budget).await?; diff --git a/crates/canopy-server/src/packs/publication/tests/ref_policy/fixture.rs b/crates/canopy-server/src/packs/publication/tests/ref_policy/fixture.rs index 6e7c7780..89ec11df 100644 --- a/crates/canopy-server/src/packs/publication/tests/ref_policy/fixture.rs +++ b/crates/canopy-server/src/packs/publication/tests/ref_policy/fixture.rs @@ -77,7 +77,8 @@ pub(super) fn attempt<'a>( *uuid::Uuid::new_v4().as_bytes(), ) .await?; - let staging = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let staging = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let ticket = staging .submit( ReadyStaging::new( diff --git a/crates/canopy-server/src/packs/publication/tests/root_completion.rs b/crates/canopy-server/src/packs/publication/tests/root_completion.rs index 628facb4..36f05fa2 100644 --- a/crates/canopy-server/src/packs/publication/tests/root_completion.rs +++ b/crates/canopy-server/src/packs/publication/tests/root_completion.rs @@ -501,13 +501,14 @@ pub(super) async fn qualify( }, expected ); - let facts = fixture.handle.query(0, 128, |db| { - let counts: (i64,i64,i64,i64,i64) = db.query_row("SELECT (SELECT count(*) FROM refs),(SELECT count(*) FROM push_responses),(SELECT count(*) FROM push_response_chunks),(SELECT count(*) FROM catalog_operations),(SELECT count(*) FROM catalog_leases)",[],|row|Ok((row.get(0)?,row.get(1)?,row.get(2)?,row.get(3)?,row.get(4)?)))?; + let token = prepared.token(); + let facts = fixture.handle.query(0, 128, move |db| { + let counts: (i64,i64,i64,i64,i64) = db.query_row("SELECT (SELECT count(*) FROM refs),(SELECT count(*) FROM push_responses),(SELECT count(*) FROM push_response_chunks),(SELECT count(*) FROM catalog_operations),(SELECT count(*) FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2)",rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt],|row|Ok((row.get(0)?,row.get(1)?,row.get(2)?,row.get(3)?,row.get(4)?)))?; Ok(serde_json::to_vec(&counts).unwrap()) }).await?; assert_eq!( serde_json::from_slice::<(i64, i64, i64, i64, i64)>(&facts)?, - (0, 0, 0, 0, 2) + (0, 0, 0, 0, 1) ); if mode == CompletionMode::WriteRevoked { edit( @@ -535,7 +536,9 @@ pub(super) async fn qualify( fixture.client().resolve(competing.evidence()).await?, Resolution::Absent )); - let known = registered.dispatch(&fixture.client(), store).await?; + let known = registered + .dispatch(&fixture.client(), store, &fixture.authority()) + .await?; assert_eq!(known.output, committed.output); assert_eq!(known.receipt, committed.receipt); assert_eq!( @@ -625,7 +628,10 @@ pub(super) async fn restored( client.resolve(competing.evidence()).await?, Resolution::Absent )); - let known = replay.registered.dispatch(&client, store).await?; + let known = replay + .registered + .dispatch(&client, store, &fixture.authority()) + .await?; assert_eq!(known.output, replay.committed.output); assert_eq!(known.receipt, replay.committed.receipt); assert_eq!(state(&handle).await?, before); diff --git a/crates/canopy-server/src/packs/publication/tests/root_dispatch.rs b/crates/canopy-server/src/packs/publication/tests/root_dispatch.rs index dee0b4d1..ceeec621 100644 --- a/crates/canopy-server/src/packs/publication/tests/root_dispatch.rs +++ b/crates/canopy-server/src/packs/publication/tests/root_dispatch.rs @@ -91,7 +91,8 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu .is_err(), "publishing native intent must not become a ref-free outcome" ); - let operation = prepared.token().operation; + let token = prepared.token(); + let operation = token.operation; let lookup = BeginRequest { repository: f.repository, operation, @@ -122,7 +123,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu let weak = Arc::downgrade(&prepared); let directory = root.to_owned(); let producer_store = store.clone(); - let producer = ticket.spawn_bound(move |session| async move { + let producer = ticket.spawn_bound(move |session, _context| async move { assert!(Arc::ptr_eq( &prepared.base.session.deadline, &session.deadline @@ -159,6 +160,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu f.repository, )?, PublicationLimits::default(), + f.publication_budget.clone(), )?; let failure = ticket .publish(&foreign, ready) @@ -174,11 +176,15 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu assert_eq!(retained.evidence_for_test(), evidence); assert_eq!(foreign.stats().await.admitted, 0); - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; p.fault_for_test(fault); let (release, wait) = tokio::sync::oneshot::channel(); let (entered, running) = tokio::sync::oneshot::channel(); - let worker = ticket.spawn_bound(move |_| async move { + let worker = ticket.spawn_bound(move |_, _context| async move { let _ = entered.send(()); wait.await.map_err(|_| StagingError::Worker)?; Ok(42u64) @@ -205,7 +211,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu let renewal = ticket.bound_renewal().ok_or("ordered root renewal")?; assert!(!close.as_ref().unwrap().is_finished()); assert_eq!(p.close_and_drain().await.len(), 1); - assert!(ticket.spawn_bound(|_| async { Ok(()) }).is_err()); + assert!(ticket.spawn_bound(|_, _context| async { Ok(()) }).is_err()); release.send(()).map_err(|_| "drained worker disappeared")?; assert_eq!( ticket @@ -245,12 +251,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu Loss::Expiry => { // Cellule's SQL deadlines use the real monotonic clock. Wait // for this shared local ceiling without shifting Tokio time. - let deadline = weak - .upgrade() - .ok_or("root custody lost")? - .base - .live_lease()? - .1; + let deadline = ticket.expire_bound_for_test()?; tokio::time::sleep_until(deadline + Duration::from_millis(1)).await; assert!( weak.upgrade() @@ -322,7 +323,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu response(observer.root_response(store).await?).await?, expected ); - assert_eq!(f.counts().await?, (0, 2)); + assert_eq!(f.counts_for(token).await?, (0, 1)); let saved = f .client() .query::(&f.target, Some(committed.receipt), lookup.clone()) @@ -336,7 +337,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, loss: Loss) -> Resu PublicationError::RootPush(InvocationError::NotStarted(_)) )); assert!(observer.root_response(store).await.is_err()); - assert_eq!(f.counts().await?, (1, 2)); + assert_eq!(f.counts_for(token).await?, (1, 1)); assert!( f.client() .query::(&f.target, None, lookup.clone()) diff --git a/crates/canopy-server/src/packs/publication/tests/root_outcome.rs b/crates/canopy-server/src/packs/publication/tests/root_outcome.rs index 49f6a678..1e45eafd 100644 --- a/crates/canopy-server/src/packs/publication/tests/root_outcome.rs +++ b/crates/canopy-server/src/packs/publication/tests/root_outcome.rs @@ -190,7 +190,11 @@ pub(super) async fn qualify(context: Context<'_>, kind: Kind, fault: u8, revoked } } let ready = ready.bind_recovery(registered, store)?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; p.fault_for_test(fault); let observer = ticket.publish(&p, ready)?; drop(observer); diff --git a/crates/canopy-server/src/packs/publication/tests/serving.rs b/crates/canopy-server/src/packs/publication/tests/serving.rs new file mode 100644 index 00000000..b585ddcf --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving.rs @@ -0,0 +1,570 @@ +//! Serving pins protect real immutable roots through worker and receipt loss. +use super::*; +use super::{initialization::empty, publishing::edit}; +use crate::packs::catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use cellule_runtime::{Committed, PreparedCommand}; +use tokio::time::{Duration, timeout}; +use tokio_util::task::TaskTracker; +mod body; +mod custody; +mod edges; +mod lifecycle; +mod pool; +mod refs; +mod selection_drain; +mod workspace; + +async fn initialize(f: &Fixture, store: Arc) -> Result { + let (prepared, root, budget) = Box::pin(empty(f, [241; 16], store.clone())).await?; + let prepared = Arc::new(prepared); + let ready = prepared.ready_initialization(identity()?).await?; + let registered = ready.persist_recovery(&store, identity()?).await?; + let InitializationReply::Initialized(fact) = ready.complete(®istered, &store).await?.output + else { + return Err("initialization not committed".into()); + }; + let admin = super::terminal_retention::maintenance(&f.handle, f.repository).await?; + assert_eq!( + registered + .ready_terminal_release(f.client(), &store, admin, identity()?) + .await? + .complete() + .await? + .output, + TerminalReleaseReply::Released + ); + drop(prepared); + super::prepare::cleaned(root.path(), &budget).await?; + Ok(*fact) +} +fn request(f: &Fixture, actor: Option<&str>, reader: u8, lease_ms: u64) -> AcquireServingRequest { + AcquireServingRequest { + repository: f.repository, + reader: [reader; 16], + actor: actor.map(str::to_owned), + lease_ms, + } +} +fn granted(reply: ServingReply) -> Result { + match reply { + ServingReply::Granted(lease) => Ok(*lease), + other => Err(format!("unexpected {other:?}").into()), + } +} +async fn acquire( + f: &Fixture, + actor: Option<&str>, + reader: u8, + lease_ms: u64, +) -> Result<( + ServingLease, + PreparedCommand, + Committed, +)> { + let command = f + .client() + .prepare_command::( + &f.target, + identity()?, + request(f, actor, reader, lease_ms), + ) + .await?; + let committed = command.clone().execute().await?; + Ok((granted(committed.output.clone())?, command, committed)) +} +async fn pin_count(f: &Fixture) -> Result { + let bytes = f + .handle + .query(0, 8, |db| { + let count: u64 = + db.query_row("SELECT count(*) FROM catalog_serving_pins", [], |row| { + row.get(0) + })?; + Ok(count.to_be_bytes().to_vec()) + }) + .await?; + Ok(u64::from_be_bytes(bytes.as_slice().try_into()?)) +} +fn context( + f: &Fixture, + store: Arc, + root: &tempfile::TempDir, + tasks: TaskTracker, +) -> Result { + Ok(ServingContext::new( + f.client(), + f.target.clone(), + f.authority(), + Arc::new(CatalogIndexes::new(store.clone(), f.format)), + Arc::new(CatalogFiles::new( + root.path(), + DiskBudget::new(64 << 20), + store, + f.format, + CatalogFileLimits::default(), + )?), + ServingReadBudget::new(4, tasks)?, + "owner".into(), + )?) +} +fn missing(f: &Fixture) -> Result { + Ok(match f.format { + ObjectFormat::Sha1 => crate::ObjectId::Sha1([7; 20]), + ObjectFormat::Sha256 => crate::ObjectId::Sha256([7; 32]), + }) +} +async fn release(f: &Fixture, pin: &ServingPin) -> Result> { + let ready = pin.ready_release(identity()?).await?; + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let ticket = coordinator.submit(ready).await?; + let state = timeout(Duration::from_secs(5), ticket.wait()).await?; + let PublicationState::Finished(Ok(PublicationOutcome::ServingRelease(result))) = state else { + return Err(format!("unexpected serving release {state:?}").into()); + }; + assert!(coordinator.close_and_drain().await.is_empty()); + Ok(result) +} + +#[tokio::test] +async fn grants_require_read_and_joint_initialization_without_allocating_namespaces() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + for (actor, reason) in [ + (Some("owner"), ServingDenial::Uninitialized), + (None, ServingDenial::Unauthorized), + (Some("other"), ServingDenial::Unauthorized), + ] { + let result = f + .client() + .command::( + &f.target, + identity()?, + request(&f, actor, 242, DEFAULT_LEASE_MS), + ) + .await; + assert!( + matches!(result, Err(InvocationError::Rejected(value)) if value.output==ServingReply::Denied(reason)) + ); + } + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let before = f.counts().await?; + let (lease, original, receipt) = acquire(&f, Some("viewer"), 243, DEFAULT_LEASE_MS).await?; + assert_eq!(lease.fact, fact); + assert_eq!(lease.token.owner, f.handle.owner_fence()); + assert_eq!(f.counts().await?, before); + assert_eq!(pin_count(&f).await?, 1); + let duplicate = f + .client() + .command::( + &f.target, + identity()?, + request(&f, Some("viewer"), 243, DEFAULT_LEASE_MS), + ) + .await; + assert!( + matches!(duplicate, Err(InvocationError::Rejected(value)) if value.output==ServingReply::Denied(ServingDenial::Conflict)) + ); + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store.clone(), &root, tasks.clone())?, + lease.token, + Some("viewer".into()), + ) + .await?; + assert_eq!( + pin.headers(Some("viewer".into()), &[missing(&f)?]).await?, + vec![None] + ); + assert_eq!( + release(&f, &pin).await?.output, + ServingReleaseReply::Released + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(pin.ready_release(identity()?).await.is_err()); + // SDK replay is immutable evidence, not a fresh serving capability. + let replay = original.execute().await?; + assert_eq!( + (replay.output, replay.receipt), + (receipt.output, receipt.receipt) + ); + assert!( + ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("viewer".into()) + ) + .await + .is_err() + ); + tasks.close(); + timeout(Duration::from_secs(5), tasks.wait()).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn anonymous_public_reads_revocation_and_token_scope_are_rechecked() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + edit(&f, "UPDATE ref_generation SET visibility='public'").await?; + let (lease, _, _) = acquire(&f, None, 244, DEFAULT_LEASE_MS).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = + ServingPin::open(context(&f, store, &root, tasks.clone())?, lease.token, None).await?; + assert_eq!(pin.headers(None, &[missing(&f)?]).await?, vec![None]); + for field in 0..5 { + let mut token = lease.token; + match field { + 0 => token.repository[15] ^= 1, + 1 => token.reader[15] ^= 1, + 2 => token.owner.epoch += 1, + 3 => token.admission_sequence += 1, + _ => token.generation += 1, + } + assert!( + f.client() + .query::(&f.target, None, ServingCheck { token, actor: None }) + .await? + .output + .is_none() + ); + } + edit(&f, "UPDATE ref_generation SET visibility='private'").await?; + assert!(matches!( + pin.headers(None, &[missing(&f)?]).await, + Err(ServingReadError::Inactive) + )); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!( + release(&f, &pin).await?.output, + ServingReleaseReply::Released + ); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn expiry_refuses_reads_and_renewal_but_retains_generation_until_drained_release() -> Result { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store.clone()).await?; + let (lease, _, _) = acquire(&f, Some("owner"), 245, 1_000).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("owner".into()), + ) + .await?; + // Fixture injection qualifies retention, not publication of a native graph. + f.install_generation(2, fact.catalog.ok_or("catalog")?, fact.refs) + .await?; + tokio::time::sleep(Duration::from_millis(1_100)).await; + assert!(matches!( + pin.headers(Some("owner".into()), &[missing(&f)?]).await, + Err(ServingReadError::Inactive) + )); + let renewal = f + .client() + .command::( + &f.target, + identity()?, + RenewServingRequest { + check: ServingCheck { + token: lease.token, + actor: Some("owner".into()), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await; + assert!( + matches!(renewal, Err(InvocationError::Rejected(value)) if value.output==ServingReply::Denied(ServingDenial::Expired)) + ); + // Expire abandoned preparation work; its floor must not mask serving retention. + edit( + &f, + "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0", + ) + .await?; + let maintenance = super::terminal_retention::maintenance(&f.handle, f.repository).await?; + f.client() + .command::(&f.target, identity()?, maintenance.clone()) + .await?; + assert_eq!(pin_count(&f).await?, 1); + assert!( + edit(&f, "DELETE FROM catalog_generations WHERE generation=1") + .await + .is_err() + ); + assert_eq!( + release(&f, &pin).await?.output, + ServingReleaseReply::Released + ); + f.client() + .command::(&f.target, identity()?, maintenance) + .await?; + let count = f + .handle + .query(0, 1, |db| { + let count: u8 = db.query_row( + "SELECT count(*) FROM catalog_generations WHERE generation=1", + [], + |row| row.get(0), + )?; + Ok(vec![count]) + }) + .await?; + assert_eq!(count, vec![0]); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[test] +fn schema_bounds_serving_pins_and_rejects_identity_replacement() -> Result { + let db = rusqlite::Connection::open_in_memory()?; + db.execute_batch("PRAGMA foreign_keys=ON")?; + db.execute_batch(SCHEMA)?; + db.execute_batch("INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(1,x'01',zeroblob(32)),(2,x'02',zeroblob(32)); INSERT INTO catalog_serving_pins VALUES(randomblob(16),zeroblob(16),1,x'0000000000000001',1,100)")?; + for sql in [ + "UPDATE catalog_serving_pins SET reader=randomblob(16)", + "UPDATE catalog_serving_pins SET incarnation=randomblob(16)", + "UPDATE catalog_serving_pins SET admission_sequence=2", + "UPDATE catalog_serving_pins SET owner_epoch=x'0000000000000002'", + "UPDATE catalog_serving_pins SET generation=2", + "UPDATE catalog_serving_pins SET expires_at_ms=99", + "INSERT OR REPLACE INTO catalog_serving_pins SELECT * FROM catalog_serving_pins", + "DELETE FROM catalog_generations WHERE generation=1", + ] { + assert!(db.execute_batch(sql).is_err(), "{sql}"); + } + db.execute_batch("UPDATE catalog_serving_pins SET expires_at_ms=101; WITH RECURSIVE n(x) AS (VALUES(2) UNION ALL SELECT x+1 FROM n WHERE x<4096) INSERT INTO catalog_serving_pins SELECT randomblob(16),zeroblob(16),x,x'0000000000000001',1,0 FROM n")?; + assert!(db.execute_batch("INSERT INTO catalog_serving_pins VALUES(randomblob(16),zeroblob(16),4097,x'0000000000000001',1,0)").is_err()); + db.execute_batch( + "DELETE FROM catalog_serving_pins; DELETE FROM catalog_generations WHERE generation=1", + )?; + Ok(()) +} + +#[tokio::test] +async fn release_uncertainty_reuses_original_command_and_receipt_after_caller_loss() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in 1..=3 { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let (lease, _, _) = acquire(&f, Some("owner"), 246, DEFAULT_LEASE_MS).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("owner".into()), + ) + .await?; + let ready = pin.ready_release(identity()?).await?; + let evidence = ready.evidence().clone(); + assert_eq!(pin.ready_release(identity()?).await?.evidence(), &evidence); + let coordinator = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + coordinator.fault_for_test(fault); + let ticket = coordinator.submit(ready).await?; + let state = timeout(Duration::from_secs(5), ticket.wait()).await?; + assert!(matches!(state, PublicationState::Uncertain(_)), "{state:?}"); + drop(ticket); + assert_eq!(pin.ready_release(identity()?).await?.evidence(), &evidence); + coordinator.fault_for_test(0); + let ticket = coordinator + .pending_serving_release(lease.token.reader) + .await + .ok_or("retained release")?; + ticket.recover().await?; + let state = timeout(Duration::from_secs(5), ticket.wait()).await?; + assert!( + matches!(state, PublicationState::Finished(Ok(PublicationOutcome::ServingRelease(ref result))) if result.output==ServingReleaseReply::Released), + "{state:?}" + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(pin.ready_release(identity()?).await.is_err()); + assert!(coordinator.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +mod blocked; + +#[tokio::test] +async fn renewal_preserves_snapshot_and_cannot_shorten_or_revive_an_existing_pin() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store).await?; + let (lease, _, _) = acquire(&f, Some("owner"), 249, DEFAULT_LEASE_MS).await?; + f.install_generation(2, fact.catalog.ok_or("catalog")?, fact.refs) + .await?; + let input = RenewServingRequest { + check: ServingCheck { + token: lease.token, + actor: Some("owner".into()), + }, + lease_ms: 1, + }; + let original = f + .client() + .prepare_command::(&f.target, identity()?, input.clone()) + .await?; + let renewed = original.clone().execute().await?; + let grant = granted(renewed.output.clone())?; + assert_eq!(grant.token, lease.token); + assert_eq!(grant.fact, fact); + assert_eq!(grant.expires_at_ms, lease.expires_at_ms); + let replay = original.execute().await?; + assert_eq!( + (replay.output, replay.receipt), + (renewed.output, renewed.receipt) + ); + edit(&f, "UPDATE repository_identity SET owner='replacement'").await?; + let result = f + .client() + .command::(&f.target, identity()?, input) + .await; + assert!( + matches!(result, Err(InvocationError::Rejected(value)) if value.output==ServingReply::Denied(ServingDenial::Unauthorized)) + ); + assert_eq!(pin_count(&f).await?, 1); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn owner_restoration_cannot_convert_an_old_pin_dto_into_fresh_serving_authority() -> Result { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let (lease, _, _) = acquire(&f, Some("owner"), 250, DEFAULT_LEASE_MS).await?; + let (runtime, handle, client) = + super::durable_recovery::restore_owner_fence(&f, lease.token.owner).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + // Use the actual restored client, not an old local handle or decoded fence. + let restored = ServingContext::new( + client.clone(), + f.target.clone(), + f.authority(), + Arc::new(CatalogIndexes::new(store.clone(), f.format)), + Arc::new(CatalogFiles::new( + root.path(), + DiskBudget::new(64 << 20), + store, + f.format, + CatalogFileLimits::default(), + )?), + ServingReadBudget::new(4, tasks.clone())?, + "owner".into(), + )?; + assert!(matches!( + ServingPin::open(restored, lease.token, Some("owner".into())).await, + Err(ServingReadError::Authority(_)) + )); + let result = client + .command::( + &f.target, + identity()?, + RenewServingRequest { + check: ServingCheck { + token: lease.token, + actor: Some("owner".into()), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await; + assert!( + matches!(result, Err(InvocationError::Rejected(value)) if value.output==ServingReply::Denied(ServingDenial::Stale)) + ); + let bytes = handle + .query(0, 8, |db| { + let count: u64 = + db.query_row("SELECT count(*) FROM catalog_serving_pins", [], |row| { + row.get(0) + })?; + Ok(count.to_be_bytes().to_vec()) + }) + .await?; + assert_eq!(u64::from_be_bytes(bytes.as_slice().try_into()?), 1); + tasks.close(); + tasks.wait().await; + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn serving_release_and_preparation_share_budgets_without_colliding_logical_ids() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let (base, _, _) = super::prepare::opened(&f, [251; 16], store.clone()).await?; + let (lease, _, _) = acquire(&f, Some("owner"), 251, DEFAULT_LEASE_MS).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("owner".into()), + ) + .await?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; + let preparation = queue.try_reserve( + Arc::new(base.session.clone()) + .ready_renew(identity()?, DEFAULT_LEASE_MS) + .await?, + )?; + assert!(matches!(preparation.state(), PublicationState::Held)); + let reader = queue.submit(pin.ready_release(identity()?).await?).await?; + assert!( + matches!(timeout(Duration::from_secs(5), reader.wait()).await?, PublicationState::Finished(Ok(PublicationOutcome::ServingRelease(ref value))) if value.output==ServingReleaseReply::Released) + ); + assert!(queue.pending([251; 16]).await.is_some()); + preparation.activate().await?; + assert!(matches!( + timeout(Duration::from_secs(5), preparation.wait()).await?, + PublicationState::Finished(Ok(PublicationOutcome::Preparation(_))) + )); + assert!(queue.close_and_drain().await.is_empty()); + assert_eq!(queue.stats().await.command_bytes, 0); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/blocked.rs b/crates/canopy-server/src/packs/publication/tests/serving/blocked.rs new file mode 100644 index 00000000..b182f77b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/blocked.rs @@ -0,0 +1,196 @@ +use super::*; +use async_trait::async_trait; +use futures_util::stream::BoxStream; +use object_store::{ + CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, + PutMultipartOptions, PutOptions, PutPayload, PutResult, +}; +use std::sync::atomic::{AtomicBool, Ordering}; +use tokio::sync::Semaphore; + +#[derive(Debug)] +pub(super) struct Gate { + store: InMemory, + pub(super) armed: AtomicBool, + pub(super) entered: Semaphore, + pub(super) proceed: Semaphore, +} +impl Gate { + pub(super) fn new() -> Self { + Self { + store: InMemory::new(), + armed: AtomicBool::new(false), + entered: Semaphore::new(0), + proceed: Semaphore::new(0), + } + } +} +impl std::fmt::Display for Gate { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "blocked serving provider") + } +} +#[async_trait] +impl ObjectStore for Gate { + async fn put_opts( + &self, + location: &Path, + payload: PutPayload, + opts: PutOptions, + ) -> object_store::Result { + self.store.put_opts(location, payload, opts).await + } + async fn put_multipart_opts( + &self, + location: &Path, + opts: PutMultipartOptions, + ) -> object_store::Result> { + self.store.put_multipart_opts(location, opts).await + } + async fn get_opts( + &self, + location: &Path, + options: GetOptions, + ) -> object_store::Result { + if self.armed.swap(false, Ordering::AcqRel) { + self.entered.add_permits(1); + self.proceed + .acquire() + .await + .expect("provider gate") + .forget(); + } + self.store.get_opts(location, options).await + } + fn delete_stream( + &self, + locations: BoxStream<'static, object_store::Result>, + ) -> BoxStream<'static, object_store::Result> { + self.store.delete_stream(locations) + } + fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { + self.store.list(prefix) + } + async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { + self.store.list_with_delimiter(prefix).await + } + async fn copy_opts( + &self, + from: &Path, + to: &Path, + options: CopyOptions, + ) -> object_store::Result<()> { + self.store.copy_opts(from, to, options).await + } +} + +#[tokio::test] +async fn canceled_observer_cannot_release_pin_while_real_provider_worker_is_suspended() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(Gate::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + initialize(&f, store.clone()).await?; + let (lease, _, _) = acquire(&f, Some("owner"), 247, DEFAULT_LEASE_MS).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store.clone(), &root, tasks.clone())?, + lease.token, + Some("owner".into()), + ) + .await?; + provider.armed.store(true, Ordering::Release); + let worker = pin.clone(); + let oid = missing(&f)?; + let observer = + tokio::spawn(async move { worker.headers(Some("owner".into()), &[oid]).await }); + timeout(Duration::from_secs(5), provider.entered.acquire()) + .await?? + .forget(); + observer.abort(); + assert!(observer.await.unwrap_err().is_cancelled()); + assert!(!tasks.is_empty()); + // Even a separately constructed context/budget cannot mint a second + // drain counter while the detached provider worker owns this pin. + assert!(matches!( + ServingPin::open( + context(&f, store.clone(), &root, tasks.clone())?, + lease.token, + Some("owner".into()) + ) + .await, + Err(ServingReadError::AlreadyOwned) + )); + + assert!( + timeout(Duration::from_millis(50), pin.ready_release(identity()?)) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + assert!(matches!( + pin.headers(Some("owner".into()), &[oid]).await, + Err(ServingReadError::Inactive) + )); + provider.proceed.add_permits(1); + tasks.close(); + timeout(Duration::from_secs(5), tasks.wait()).await?; + assert_eq!( + release(&f, &pin).await?.output, + ServingReleaseReply::Released + ); + assert_eq!(pin_count(&f).await?, 0); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn revocation_during_provider_io_discards_result_without_abandoning_retention() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let provider = Arc::new(Gate::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let (lease, _, _) = acquire(&f, Some("viewer"), 248, DEFAULT_LEASE_MS).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store.clone(), &root, tasks.clone())?, + lease.token, + Some("viewer".into()), + ) + .await?; + provider.armed.store(true, Ordering::Release); + let worker = pin.clone(); + let oid = missing(&f)?; + let observer = tokio::spawn(async move { worker.headers(Some("viewer".into()), &[oid]).await }); + timeout(Duration::from_secs(5), provider.entered.acquire()) + .await?? + .forget(); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + assert!( + timeout(Duration::from_millis(50), pin.close_and_drain()) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + provider.proceed.add_permits(1); + assert!(matches!( + timeout(Duration::from_secs(5), observer).await??, + Err(ServingReadError::Inactive) + )); + assert_eq!( + release(&f, &pin).await?.output, + ServingReleaseReply::Released + ); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/body.rs b/crates/canopy-server/src/packs/publication/tests/serving/body.rs new file mode 100644 index 00000000..4ddb0371 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/body.rs @@ -0,0 +1,261 @@ +//! Actual native packs/metadata with trusted catalog installation to isolate +//! serving semantics. This is not end-to-end producer publication qualification. +use super::*; +use crate::packs::{ + catalog::{CatalogSnapshot, StoredCatalog}, + metadata::tests::limits, + verification::physical::tests::{Prepared, prepared_for_store}, +}; +use object_store::ObjectStore; + +pub(super) async fn catalog( + f: &Fixture, + provider: Arc, +) -> Result<(Prepared, StoredCatalog)> { + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + let (base, _, _) = super::super::prepare::opened(f, [71; 16], store.clone()).await?; + let native = + prepared_for_store(f.format, 32, base.context().operation, provider, store).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segments) = + super::super::prepare::physical(&native, root.path(), budget.clone()).await?; + builder.begin_pack(witness)?; + for segment in segments { + builder.add_segment(segment).await?; + } + builder.finish_pack().await?; + let prepared = builder.finish().await?; + let stored = prepared.catalog(); + drop(prepared); + super::super::prepare::cleaned(root.path(), &budget).await?; + let fact = initialize(f, native.store.clone()).await?; + f.install_generation(2, stored, fact.refs).await?; + Ok((native, stored)) +} +pub(super) fn serving_context( + f: &Fixture, + store: Arc, + root: &tempfile::TempDir, + tasks: TaskTracker, +) -> Result<(ServingContext, Arc)> { + let files = CatalogFiles::new( + root.path(), + DiskBudget::new(64 << 20), + store.clone(), + f.format, + CatalogFileLimits::default(), + )? + .with_native( + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ); + let files = Arc::new(files); + Ok(( + ServingContext::new( + f.client(), + f.target.clone(), + f.authority(), + Arc::new(CatalogIndexes::new(store, f.format)), + files.clone(), + ServingReadBudget::new(32, tasks)?, + "owner".into(), + )?, + files, + )) +} + +#[tokio::test] +async fn certified_bodies_share_one_pack_across_parallel_readers_and_recheck_cached_access() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (native, _) = catalog(&f, Arc::new(InMemory::new())).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("viewer".into())).await?; + let ids: Vec<_> = native.fixture.objects.keys().copied().collect(); + let results = + futures_util::future::join_all(ids.iter().take(12).map(|oid| view.body(*oid, 1 << 20))) + .await; + for (oid, body) in ids.iter().zip(results) { + let body = body?.ok_or("body missing")?; + let expected = native.fixture.objects[oid].0; + assert_eq!(crate::object_id(format, expected.kind, &body), *oid); + assert_eq!(blake3::hash(&body).as_bytes(), &expected.digest); + } + for (oid, (expected, _)) in &native.fixture.objects { + let body = view.body(*oid, 1 << 20).await?.ok_or("body missing")?; + assert_eq!(body.len() as u64, expected.size); + } + let stats = files.native_stats()?.ok_or("native stats")?; + assert_eq!(stats.downloaded_files, 1); + assert_eq!(stats.open_files, 1); + assert_eq!(stats.cached_files, 1); + assert!(stats.cache_hits >= 12); + assert_eq!(view.body(missing(&f)?, 1024).await?, None); + let blob = native + .fixture + .objects + .values() + .find(|(o, _)| o.kind == crate::ObjectKind::Blob) + .ok_or("blob")? + .0; + assert!(matches!( + view.body(blob.oid, 1).await, + Err(ServingReadError::TooLarge) + )); + assert!(matches!( + view.body(blob.oid, 65 << 20).await, + Err(ServingReadError::TooLarge) + )); + assert_eq!( + files + .native_stats()? + .ok_or("native stats")? + .downloaded_files, + 1 + ); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + assert!(matches!( + view.body(blob.oid, 1024).await, + Err(ServingReadError::Inactive) + )); + drop(view); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn cached_pack_never_exposes_objects_absent_from_the_selected_catalog() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (native, _) = catalog(&f, Arc::new(InMemory::new())).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, _) = serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let old = pool.snapshot(Some("owner".into())).await?; + let oid = *native.fixture.objects.keys().next().ok_or("object")?; + assert!(old.body(oid, 1 << 20).await?.is_some()); + let directory = + crate::packs::directory::snapshot::DirectorySnapshot::empty(f.repository, format) + .upload(&native.store, [181; 16]) + .await?; + let empty = CatalogSnapshot { + directory, + sources: None, + } + .upload(&native.store, [182; 16]) + .await?; + f.install_generation(3, empty, old.fact().refs).await?; + let current = pool.snapshot(Some("owner".into())).await?; + assert_eq!(current.body(oid, 1 << 20).await?, None); + assert!(old.body(oid, 1 << 20).await?.is_some()); + drop((old, current)); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn blocked_native_pack_download_retains_pin_after_observer_cancellation() -> Result { + use std::sync::atomic::Ordering; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = catalog(&f, provider.clone()).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, _) = serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("owner".into())).await?; + let oid = *native.fixture.objects.keys().next().ok_or("object")?; + // Warm all metadata; the armed provider suspension is actual pack I/O. + assert!(view.headers(&[oid]).await?[0].is_some()); + provider.armed.store(true, Ordering::Release); + let observer = tokio::spawn(async move { view.body(oid, 1 << 20).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + observer.abort(); + assert!(observer.await.err().ok_or("observer")?.is_cancelled()); + assert!(!pool.quiesce().await?); + let mut drain = tokio::spawn({ + let pool = pool.clone(); + async move { pool.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + provider.proceed.add_permits(1); + timeout(Duration::from_secs(8), drain).await??; + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn native_download_drain_finishes_but_body_is_refused_after_access_revocation() -> Result { + use std::sync::atomic::Ordering; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = catalog(&f, provider.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("viewer".into())).await?; + let oid = *native.fixture.objects.keys().next().ok_or("object")?; + assert!(view.headers(&[oid]).await?[0].is_some()); + provider.armed.store(true, Ordering::Release); + let observer = tokio::spawn(async move { view.body(oid, 1 << 20).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + provider.proceed.add_permits(1); + assert!(matches!( + timeout(Duration::from_secs(8), observer).await??, + Err(ServingReadError::Inactive) + )); + let authorized = pool.snapshot(Some("owner".into())).await?; + assert!(authorized.body(oid, 1 << 20).await?.is_some()); + assert_eq!( + files + .native_stats()? + .ok_or("native stats")? + .downloaded_files, + 1 + ); + drop(authorized); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/custody.rs b/crates/canopy-server/src/packs/publication/tests/serving/custody.rs new file mode 100644 index 00000000..e63ee2bf --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/custody.rs @@ -0,0 +1,618 @@ +//! Serving uses the same exact intent protocol without gaining write custody. +use super::*; +use cellule_runtime::{PendingMutation, Resolution}; + +fn queue(f: &Fixture) -> Result { + Ok(PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?) +} +async fn ready(f: &Fixture, reader: u8, actor: &str) -> Result { + let mut request = f.begin([reader; 16]); + request.actor = actor.into(); + Ok(ReadyServingCommand::acquire( + f.client(), + f.target.clone(), + request, + identity()?, + f.authority(), + ) + .await?) +} +async fn result(ticket: &PublicationTicket) -> Result> { + match timeout(Duration::from_secs(10), ticket.wait()).await? { + PublicationState::Finished(Ok(PublicationOutcome::ServingCommand(value))) => Ok(value), + other => Err(format!("serving command: {other:?}").into()), + } +} +async fn saved(f: &Fixture, reader: u8) -> Result { + Ok(RegisteredCustody::load_for( + &f.client(), + &f.target, + CustodyPurpose::Serving, + [reader; 16], + ) + .await? + .ok_or("serving intent missing")?) +} +async fn expired(evidence: &PendingMutation) -> Result { + let now = sql::now(0)?; + if now <= evidence.identity().expires_at_ms { + tokio::time::sleep(Duration::from_millis(u64::try_from( + evidence.identity().expires_at_ms - now + 1, + )?)) + .await; + } + Ok(()) +} + +#[tokio::test] +async fn readonly_serving_and_creating_same_id_keep_distinct_originals_and_no_namespace() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let creating = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginPreparation(f.begin([220; 16])), + identity()?, + ) + .await? + .register(&f.client(), identity()?) + .await?; + let before = f.counts().await?; + let q = queue(&f)?; + let serving = ready(&f, 220, "viewer").await?; + let evidence = serving.evidence().clone(); + assert_ne!(&evidence, creating.evidence()); + let committed = result(&q.submit(serving).await?).await?; + let lease = granted(committed.output.clone())?; + assert_eq!(lease.fact, fact); + assert_eq!( + lease.token.admission_sequence, + committed.receipt.commit_sequence + ); + assert_eq!(f.counts().await?, before); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(saved(&f, 220).await?.evidence(), &evidence); + assert_eq!( + saved(&f, 220).await?.recover_serving(&f.client()).await?, + committed + ); + assert_eq!( + RegisteredCustody::load_latest(&f.client(), &f.target, [220; 16]) + .await? + .ok_or("creating intent lost")? + .evidence(), + creating.evidence() + ); + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("viewer".into()), + ) + .await?; + assert_eq!( + release(&f, &pin).await?.output, + ServingReleaseReply::Released + ); + // Historical knowledge is unchanged by physical release. + assert_eq!( + saved(&f, 220).await?.recover_serving(&f.client()).await?, + committed + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn acquisition_six_transport_faults_recover_exact_original_after_observer_loss() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in 1..=6 { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let before = f.counts().await?; + let q = queue(&f)?; + let original = ready(&f, 221, "owner").await?; + let evidence = original.evidence().clone(); + q.fault_for_test(fault); + let ticket = q.submit(original).await?; + assert!( + matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + PublicationState::Uncertain(_) + ), + "fault {fault}" + ); + assert_eq!(pin_count(&f).await?, u64::from(matches!(fault, 2 | 3))); + assert_eq!( + q.stats().await.command_bytes, + crate::packs::publication::custody::RESERVATION + ); + assert_eq!(q.stats().await.foreground, 1); + drop(ticket); + let ticket = q + .pending_serving_command([221; 16]) + .await + .ok_or("original lost")?; + assert_eq!(q.close_and_drain().await.len(), 1); + ticket.recover().await?; + let committed = result(&ticket).await?; + let lease = granted(committed.output.clone())?; + let saved = saved(&f, 221).await?; + assert_eq!(saved.evidence(), &evidence); + assert_eq!(saved.recover_serving(&f.client()).await?, committed); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(f.counts().await?, before); + assert_eq!(q.stats().await.command_bytes, 0); + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("owner".into()), + ) + .await?; + release(&f, &pin).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn renewal_retains_physical_drain_guard_across_all_uncertain_transports() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in 1..=6 { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let lease = granted( + result(&q.submit(ready(&f, 222, "owner").await?).await?) + .await? + .output, + )?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("owner".into()), + ) + .await?; + let renewal = pin + .ready_renew("owner".into(), [17; 32], identity()?, DEFAULT_LEASE_MS) + .await?; + let evidence = renewal.evidence().clone(); + let held = q.try_reserve(renewal)?; + let observer = held.clone(); + let observing = tokio::spawn(async move { observer.wait().await }); + tokio::task::yield_now().await; + observing.abort(); + assert!(observing.await.unwrap_err().is_cancelled()); + drop(held); + let closing_pin = pin.clone(); + let closing = tokio::spawn(async move { closing_pin.close_and_drain().await }); + tokio::task::yield_now().await; + assert!(!closing.is_finished()); + let held = q + .pending_serving_command([222; 16]) + .await + .ok_or("held renewal lost")?; + assert_eq!(q.close_and_drain().await.len(), 1); + q.fault_for_test(fault); + held.activate().await?; + assert!(matches!( + timeout(Duration::from_secs(10), held.wait()).await?, + PublicationState::Uncertain(_) + )); + assert!(!closing.is_finished()); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!( + q.stats().await.command_bytes, + crate::packs::publication::custody::RESERVATION + ); + held.recover().await?; + let renewed = granted(result(&held).await?.output)?; + assert_eq!(renewed.token, lease.token); + assert_eq!(renewed.fact, lease.fact); + assert!(renewed.expires_at_ms >= lease.expires_at_ms); + assert_eq!(saved(&f, 222).await?.evidence(), &evidence); + timeout(Duration::from_secs(5), closing).await??; + assert!( + pin.ready_renew("owner".into(), [17; 32], identity()?, 1) + .await + .is_err() + ); + release(&f, &pin).await?; + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn discarded_unexecuted_renewal_releases_guard_and_preserves_acquisition_head() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let acquisition = result(&q.submit(ready(&f, 223, "owner").await?).await?).await?; + let original = saved(&f, 223).await?.evidence().clone(); + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + granted(acquisition.output)?.token, + Some("owner".into()), + ) + .await?; + let renewal = pin + .ready_renew("owner".into(), [17; 32], identity()?, 1) + .await?; + let evidence = renewal.evidence().clone(); + let held = q.try_reserve(renewal)?; + let closing_pin = pin.clone(); + let closing = tokio::spawn(async move { closing_pin.close_and_drain().await }); + tokio::task::yield_now().await; + assert!(!closing.is_finished()); + held.discard_held().await?; + timeout(Duration::from_secs(5), closing).await??; + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + assert_eq!(saved(&f, 223).await?.evidence(), &original); + release(&f, &pin).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn late_serving_phase_fault_rolls_back_pin_and_sdk_acceptance_before_exact_retry() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for ignore in [false, true] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store).await?; + let prepared = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::AcquireServing(f.begin([224; 16])), + identity()?, + ) + .await?; + let registered = prepared.register(&f.client(), identity()?).await?; + let evidence = registered.evidence().clone(); + let raised = if ignore { + "IGNORE" + } else { + "ABORT,'late serving phase fault'" + }; + edit(&f, &format!("CREATE TRIGGER serving_phase_fault BEFORE UPDATE OF phase ON catalog_custody_commands WHEN NEW.purpose=1 BEGIN SELECT RAISE({raised}); END")).await?; + assert!(matches!( + registered.recover_serving(&f.client()).await, + Err(InvocationError::NotStarted(_)) + )); + assert_eq!(pin_count(&f).await?, 0); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + assert!(!saved(&f, 224).await?.settled()); + edit(&f, "DROP TRIGGER serving_phase_fault").await?; + let committed = registered.recover_serving(&f.client()).await?; + assert_eq!( + granted(committed.output.clone())?.token.admission_sequence, + committed.receipt.commit_sequence + ); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(registered.recover_serving(&f.client()).await?, committed); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn revoked_read_is_recorded_as_original_denial_and_never_rewritten_by_restored_access() +-> Result { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let mut request = f.begin([225; 16]); + request.actor = "viewer".into(); + let prepared = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::AcquireServing(request), + identity()?, + ) + .await?; + let registered = prepared.register(&f.client(), identity()?).await?; + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + let Err(InvocationError::Rejected(first)) = registered.recover_serving(&f.client()).await + else { + return Err("revoked read was not recorded as denied".into()); + }; + assert_eq!( + first.output, + ServingReply::Denied(ServingDenial::Unauthorized) + ); + assert!(saved(&f, 225).await?.settled()); + assert_eq!(pin_count(&f).await?, 0); + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = queue(&f)?; + let restored = + ReadyServingCommand::restore(f.client(), f.target.clone(), [225; 16], f.authority()) + .await?; + assert_eq!(restored.evidence(), registered.evidence()); + let state = timeout(Duration::from_secs(5), q.submit(restored).await?.wait()).await?; + assert!(matches!(state, PublicationState::Finished(Err(ref error)) + if matches!(&**error, PublicationError::ServingCommand(InvocationError::Rejected(value)) if **value==*first))); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn cold_owner_restoration_preserves_grant_receipt_but_refuses_old_physical_authority() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let original = ReadyServingCommand::acquire( + f.client(), + f.target.clone(), + f.begin([226; 16]), + mutation, + f.authority(), + ) + .await?; + let committed = result(&q.submit(original).await?).await?; + let lease = granted(committed.output.clone())?; + let evidence = saved(&f, 226).await?.evidence().clone(); + assert!(q.close_and_drain().await.is_empty()); + let (runtime, handle, client) = + super::super::durable_recovery::restore_owner_fence(&f, lease.token.owner).await?; + assert_ne!(handle.owner_fence(), lease.token.owner); + expired(&evidence).await?; + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Expired + )); + let recovered = + RegisteredCustody::load_for(&client, &f.target, CustodyPurpose::Serving, [226; 16]) + .await? + .ok_or("durable serving history lost")?; + assert_eq!(recovered.evidence(), &evidence); + assert_eq!(recovered.recover_serving(&client).await?, committed); + let cold_q = queue(&f)?; + let restored = ReadyServingCommand::restore( + client.clone(), + f.target.clone(), + [226; 16], + f.authority(), + ) + .await?; + assert_eq!(result(&cold_q.submit(restored).await?).await?, committed); + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let ctx = ServingContext::new( + client, + f.target.clone(), + f.authority(), + Arc::new(CatalogIndexes::new(store.clone(), format)), + Arc::new(CatalogFiles::new( + root.path(), + DiskBudget::new(64 << 20), + store, + format, + CatalogFileLimits::default(), + )?), + ServingReadBudget::new(4, tasks.clone())?, + "owner".into(), + )?; + assert!(matches!( + ServingPin::open(ctx, lease.token, Some("owner".into())).await, + Err(ServingReadError::Authority(PreparationBaseError::Inactive)) + )); + handle + .query(0, 8, |db| { + assert_eq!( + db.query_row("SELECT count(*) FROM catalog_serving_pins", [], |row| row + .get::<_, u64>( + 0 + ))?, + 1 + ); + Ok(Vec::new()) + }) + .await?; + assert!(cold_q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn scanner_retires_both_purposes_with_same_id_without_removing_accepted_serving_root() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let live = granted( + result(&q.submit(ready(&f, 228, "owner").await?).await?) + .await? + .output, + )?; + let mut records = Vec::new(); + for purpose in [CustodyPurpose::Creating, CustodyPurpose::Serving] { + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 200; + let action = if purpose == CustodyPurpose::Creating { + CustodyAction::BeginPreparation(f.begin([227; 16])) + } else { + CustodyAction::AcquireServing(f.begin([227; 16])) + }; + let registered = PreparedCustody::prepare(&f.client(), &f.target, action, mutation) + .await? + .register(&f.client(), identity()?) + .await?; + records.push((purpose, registered)); + } + for (_, record) in &records { + expired(record.evidence()).await?; + } + let service = CustodySupervisor::start( + f.client(), + f.target.clone(), + q.clone(), + f.scans(RecoveryScanLimits { + page: 1, + interval: Duration::from_millis(10), + }), + f.authority(), + )?; + timeout(Duration::from_secs(10), async { + loop { + let mut complete = true; + for (purpose, _) in &records { + complete &= + RegisteredCustody::load_for(&f.client(), &f.target, *purpose, [227; 16]) + .await? + .ok_or("scope disappeared")? + .stop_fact() + .is_some(); + } + if complete { + return Ok::<_, Box>(()); + } + tokio::task::yield_now().await; + } + }) + .await??; + let stats = service.shutdown().await?; + assert_eq!(stats.submitted, 2); + assert_eq!(stats.failures, 0); + for (purpose, original) in records { + let saved = RegisteredCustody::load_for(&f.client(), &f.target, purpose, [227; 16]) + .await? + .ok_or("stopped scope lost")?; + assert_eq!(saved.evidence(), original.evidence()); + assert!(!saved.settled()); + assert!(saved.closed()); + assert!(matches!(saved.recover(&f.client()).await, + Err(InvocationError::Pending(value)) if *value==*original.evidence())); + } + assert_eq!(pin_count(&f).await?, 1); + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pin = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + live.token, + Some("owner".into()), + ) + .await?; + release(&f, &pin).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn serving_journal_identity_is_immutable_and_pending_quota_is_shared() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let prepared = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::AcquireServing(f.begin([229; 16])), + identity()?, + ) + .await?; + let registered = prepared.register(&f.client(), identity()?).await?; + for statement in [ + "UPDATE catalog_custody_commands SET purpose=0 WHERE purpose=1", + "INSERT OR REPLACE INTO catalog_custody_commands SELECT * FROM catalog_custody_commands", + "INSERT INTO catalog_custody_commands(purpose,operation,step,incarnation,request_id,intent) SELECT 0,operation,step,incarnation,request_id,intent FROM catalog_custody_commands", + "INSERT INTO catalog_custody_commands(operation,step,incarnation,request_id,intent) VALUES(zeroblob(16),0,zeroblob(16),zeroblob(16),x'01')", + "DELETE FROM catalog_custody_commands", + ] { + assert!(edit(&f, statement).await.is_err(), "{statement}"); + } + assert_eq!(saved(&f, 229).await?.evidence(), registered.evidence()); + assert!( + RegisteredCustody::load_latest(&f.client(), &f.target, [229; 16]) + .await? + .is_none() + ); + // Trusted invalid heads qualify the bounded quota probe, not recovery. + edit(&f, "WITH RECURSIVE n(x) AS (VALUES(1) UNION ALL SELECT x+1 FROM n WHERE x<1023) INSERT INTO catalog_custody_commands(purpose,operation,step,incarnation,request_id,intent) SELECT 0,CAST(printf('%016d',x) AS BLOB),0,zeroblob(16),CAST(printf('%016d',x) AS BLOB),x'01' FROM n").await?; + for action in [ + CustodyAction::AcquireServing(f.begin([230; 16])), + CustodyAction::BeginPreparation(f.begin([230; 16])), + ] { + let original = + PreparedCustody::prepare(&f.client(), &f.target, action, identity()?).await?; + assert!(matches!(original.register(&f.client(), identity()?).await, + Err(CustodyError::Registration(error)) if matches!(&*error, + InvocationError::Rejected(value) if value.output==RootRecoveryReply::Denied(PreparationDenial::Capacity)))); + assert!(matches!( + f.client().resolve(original.evidence()).await?, + Resolution::Absent + )); + } + assert_eq!(f.counts().await?, (0, 0)); + assert_eq!(pin_count(&f).await?, 0); + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/edges.rs b/crates/canopy-server/src/packs/publication/tests/serving/edges.rs new file mode 100644 index 00000000..2f65a439 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/edges.rs @@ -0,0 +1,96 @@ +//! Typed graph reads retain the same owned lifetime as headers and bodies. +use super::*; +use std::sync::atomic::Ordering; + +#[tokio::test] +async fn canceled_edge_page_keeps_generation_until_actual_provider_drain() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = super::body::catalog(&f, provider.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let q = super::pool::queue(&f)?; + let (context, _) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(context, q.clone(), ServingPoolLimits::default())?; + let snapshot = pool.snapshot(Some("owner".into())).await?; + let id = *native.fixture.objects.keys().next().ok_or("object")?; + provider.armed.store(true, Ordering::Release); + let observer = tokio::spawn(async move { snapshot.edges_page(&[id], None).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + observer.abort(); + assert!(observer.await.err().ok_or("observer")?.is_cancelled()); + assert!(!pool.quiesce().await?); + let mut drain = tokio::spawn({ + let pool = pool.clone(); + async move { pool.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + provider.proceed.add_permits(1); + timeout(Duration::from_secs(8), drain).await??; + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn edge_pages_discard_revoked_in_flight_results_and_recheck_cache_hits() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = super::body::catalog(&f, provider.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let q = super::pool::queue(&f)?; + let (context, _) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(context, q.clone(), ServingPoolLimits::default())?; + let viewer = pool.snapshot(Some("viewer".into())).await?; + let id = native + .fixture + .objects + .iter() + .find(|(_, (_, edges))| !edges.is_empty()) + .ok_or("object with edges")? + .0; + let id = *id; + provider.armed.store(true, Ordering::Release); + let observer = tokio::spawn({ + let viewer = viewer.clone(); + async move { viewer.edges_page(&[id], None).await } + }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + provider.proceed.add_permits(1); + assert!(matches!( + timeout(Duration::from_secs(8), observer).await??, + Err(ServingReadError::Inactive) + )); + let owner = pool.snapshot(Some("owner".into())).await?; + assert!(!owner.edges_page(&[id], None).await?.edges.is_empty()); + assert!(matches!( + viewer.edges_page(&[id], None).await, + Err(ServingReadError::Inactive) + )); + drop((owner, viewer)); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/lifecycle.rs b/crates/canopy-server/src/packs/publication/tests/serving/lifecycle.rs new file mode 100644 index 00000000..0aaab02d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/lifecycle.rs @@ -0,0 +1,595 @@ +//! Accepted ownership survives actual command, lease and observer boundaries. +use super::*; +use cellule_runtime::Resolution; +mod restarts; + +fn queue(f: &Fixture) -> Result { + Ok(PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?) +} +fn input(f: &Fixture, reader: u8, actor: &str, lease_ms: u64) -> BeginRequest { + let mut input = f.begin([reader; 16]); + input.actor = actor.into(); + input.lease_ms = lease_ms; + input +} +fn shared_context( + f: &Fixture, + store: Arc, + root: &tempfile::TempDir, + budget: ServingReadBudget, +) -> Result { + Ok(ServingContext::new( + f.client(), + f.target.clone(), + f.authority(), + Arc::new(CatalogIndexes::new(store.clone(), f.format)), + Arc::new(CatalogFiles::new( + root.path(), + DiskBudget::new(64 << 20), + store, + f.format, + CatalogFileLimits::default(), + )?), + budget, + "owner".into(), + )?) +} +async fn ready(owner: &ServingOwner) -> Result { + Ok(timeout(Duration::from_secs(8), async { + loop { + let stats = owner.stats(); + if stats.phase == ServingOwnerPhase::Ready { + break stats; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?) +} +async fn zero_pins(f: &Fixture) -> Result { + timeout(Duration::from_secs(8), async { + while pin_count(f).await.unwrap() != 0 { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + Ok(()) +} +async fn command(ticket: &PublicationTicket) -> Result> { + match timeout(Duration::from_secs(8), ticket.wait()).await? { + PublicationState::Finished(Ok(PublicationOutcome::ServingCommand(value))) => Ok(value), + state => Err(format!("serving command {state:?}").into()), + } +} +async fn original( + f: &Fixture, + reader: u8, + actor: &str, + lease_ms: u64, +) -> Result { + Ok(ReadyServingCommand::acquire( + f.client(), + f.target.clone(), + input(f, reader, actor, lease_ms), + identity()?, + f.authority(), + ) + .await?) +} + +#[tokio::test] +async fn accepted_local_handoff_retains_expired_revoked_grant_without_granting_read() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = queue(&f)?; + let request = original(&f, 212, "viewer", 1_000).await?; + let grant = granted( + command(&q.submit(request.dispatch_copy()).await?) + .await? + .output, + )?; + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + tokio::time::sleep(Duration::from_millis(1_100)).await; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let budget = ServingReadBudget::new(4, tasks.clone())?; + budget.close(); + let pin = request + .retain_acquisition(shared_context(&f, store.clone(), &root, budget)?) + .await?; + assert_eq!(pin.token(), grant.token); + assert_eq!(pin.fact(), fact); + assert!(matches!( + pin.headers(Some("viewer".into()), &[missing(&f)?]).await, + Err(ServingReadError::Inactive) + )); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!( + release(&f, &pin).await?.output, + ServingReleaseReply::Released + ); + assert!( + request + .retain_acquisition(context(&f, store, &root, tasks.clone())?) + .await + .is_err() + ); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn missing_denied_and_restored_originals_cannot_mint_physical_handoff() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pending = original(&f, 214, "owner", DEFAULT_LEASE_MS).await?; + assert!( + pending + .retain_acquisition(context(&f, store.clone(), &root, tasks.clone())?) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 0); + q.fault_for_test(1); + let ticket = q.submit(pending.dispatch_copy()).await?; + assert!(matches!( + ticket.wait().await, + PublicationState::Uncertain(_) + )); + assert!( + pending + .retain_acquisition(context(&f, store.clone(), &root, tasks.clone())?) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 0); + ticket.recover().await?; + let accepted = command(&ticket).await?; + let restored = + ReadyServingCommand::restore(f.client(), f.target.clone(), [214; 16], f.authority()) + .await?; + assert_eq!(restored.evidence(), pending.evidence()); + assert!(matches!( + restored + .retain_acquisition(context(&f, store.clone(), &root, tasks.clone())?) + .await, + Err(ServingReadError::Context) + )); + let pin = pending + .retain_acquisition(context(&f, store.clone(), &root, tasks.clone())?) + .await?; + assert_eq!(pin.token(), granted(accepted.output)?.token); + assert!(matches!( + pending + .retain_acquisition(context(&f, store.clone(), &root, tasks.clone())?) + .await, + Err(ServingReadError::AlreadyOwned) + )); + let denied = original(&f, 215, "other", DEFAULT_LEASE_MS).await?; + assert!(matches!( + q.submit(denied.dispatch_copy()).await?.wait().await, + PublicationState::Finished(Err(_)) + )); + assert!( + denied + .retain_acquisition(context(&f, store, &root, tasks.clone())?) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + release(&f, &pin).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn handoff_uses_original_acquisition_ordinal_after_a_later_renewal() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let original = original(&f, 216, "owner", DEFAULT_LEASE_MS).await?; + let accepted = command(&q.submit(original.dispatch_copy()).await?).await?; + let pin = original + .retain_acquisition(context(&f, store.clone(), &root, tasks.clone())?) + .await?; + let renewal = pin + .ready_renew( + "owner".into(), + f.begin([216; 16]).request_digest, + identity()?, + DEFAULT_LEASE_MS, + ) + .await?; + let renewed = command(&q.submit(renewal.dispatch_copy()).await?).await?; + assert!(renewed.receipt.commit_sequence > accepted.receipt.commit_sequence); + assert!(matches!( + renewal + .retain_acquisition(context(&f, store.clone(), &root, tasks.clone())?) + .await, + Err(ServingReadError::Context) + )); + drop(renewal); + drop(pin); + let retained = original + .retain_acquisition(context(&f, store, &root, tasks.clone())?) + .await?; + assert_eq!( + retained.token().admission_sequence, + accepted.receipt.commit_sequence + ); + assert_eq!(retained.fact(), fact); + release(&f, &retained).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn automatic_owner_recovers_all_six_fault_modes_after_snapshot_observer_loss() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in 1..=6 { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (dispatch, entered) = q.pause_for_test().await; + q.fault_for_test(fault); + let owner = ServingOwner::start( + context(&f, store, &root, tasks.clone())?, + q.clone(), + input(&f, 217, "owner", DEFAULT_LEASE_MS), + identity()?, + ) + .await?; + let observed = owner.clone(); + let observer = + tokio::spawn(async move { observed.snapshot(Some("owner".into())).await }); + timeout(Duration::from_secs(8), entered).await??; + observer.abort(); + assert!( + observer + .await + .err() + .ok_or("observer completed")? + .is_cancelled() + ); + dispatch.send(()).map_err(|_| "dispatcher lost")?; + let stats = ready(&owner).await?; + assert_eq!(stats.token.ok_or("token")?.reader, [217; 16]); + let saved = RegisteredCustody::load_for( + &f.client(), + &f.target, + CustodyPurpose::Serving, + [217; 16], + ) + .await? + .ok_or("original missing")?; + assert_eq!( + saved + .recover_serving(&f.client()) + .await? + .receipt + .commit_sequence, + stats.token.unwrap().admission_sequence + ); + let snapshot = owner.snapshot(Some("owner".into())).await?; + assert_eq!(snapshot.headers(&[missing(&f)?]).await?, vec![None]); + drop(snapshot); + assert_eq!( + timeout(Duration::from_secs(8), owner.close_and_drain()) + .await? + .phase, + ServingOwnerPhase::Released + ); + assert_eq!(pin_count(&f).await?, 0); + assert_eq!(q.stats().await.command_bytes, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn closed_owner_keeps_borrowed_generation_renewing_until_last_snapshot_clone_drops() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let owner = ServingOwner::start( + context(&f, store, &root, tasks.clone())?, + q.clone(), + input(&f, 218, "owner", 1_000), + identity()?, + ) + .await?; + ready(&owner).await?; + let snapshot = owner.snapshot(Some("owner".into())).await?; + let clone = snapshot.clone(); + let token = owner.stats().token.ok_or("token")?; + owner.close(); + assert!(matches!( + owner.snapshot(Some("owner".into())).await, + Err(ServingReadError::Inactive) + )); + tokio::time::sleep(Duration::from_millis(1_500)).await; + timeout(Duration::from_secs(8), async { + while owner.stats().renewals < 2 { + assert_eq!( + owner.stats().phase, + ServingOwnerPhase::Ready, + "{:?}", + owner.stats() + ); + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert!(owner.stats().renewals >= 2, "{:?}", owner.stats()); + assert_eq!(owner.stats().token, Some(token)); + assert_eq!(snapshot.fact(), fact); + assert_eq!(snapshot.headers(&[missing(&f)?]).await?, vec![None]); + drop(snapshot); + assert_eq!(pin_count(&f).await?, 1); + drop(clone); + assert_eq!( + timeout(Duration::from_secs(8), owner.close_and_drain()) + .await? + .phase, + ServingOwnerPhase::Released + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn last_handle_drop_drains_after_borrow_and_joins_uncertain_release() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let owner = ServingOwner::start( + context(&f, store, &root, tasks.clone())?, + q.clone(), + input(&f, 219, "owner", DEFAULT_LEASE_MS), + identity()?, + ) + .await?; + ready(&owner).await?; + let snapshot = owner.snapshot(Some("owner".into())).await?; + let drained = owner.drain_observer(); + drop(owner); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(snapshot.headers(&[missing(&f)?]).await?, vec![None]); + q.fault_for_test(3); + drop(snapshot); + assert_eq!( + timeout(Duration::from_secs(8), drained.wait()).await?.phase, + ServingOwnerPhase::Released + ); + zero_pins(&f).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn denied_release_stays_owned_until_current_admin_can_authentically_release() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let owner = ServingOwner::start( + context(&f, store, &root, tasks.clone())?, + q.clone(), + input(&f, 220, "owner", DEFAULT_LEASE_MS), + identity()?, + ) + .await?; + ready(&owner).await?; + let (dispatch, entered) = q.pause_for_test().await; + owner.close(); + timeout(Duration::from_secs(8), entered).await??; + edit(&f, "UPDATE repository_identity SET owner='other'").await?; + dispatch.send(()).map_err(|_| "dispatcher lost")?; + timeout(Duration::from_secs(8), async { + while owner.stats().last_error.is_none() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert_eq!(pin_count(&f).await?, 1); + let closing = owner.clone(); + let observer = tokio::spawn(async move { closing.close_and_drain().await }); + tokio::time::sleep(Duration::from_millis(200)).await; + assert!(!observer.is_finished()); + observer.abort(); + assert!(observer.await.err().ok_or("closed early")?.is_cancelled()); + edit(&f, "UPDATE repository_identity SET owner='owner'").await?; + assert_eq!( + timeout(Duration::from_secs(8), owner.close_and_drain()) + .await? + .phase, + ServingOwnerPhase::Released + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn closed_read_budget_refuses_new_work_but_preserves_existing_owner_cleanup() -> Result { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let budget = ServingReadBudget::new(4, tasks.clone())?; + let context = shared_context(&f, store, &root, budget.clone())?; + let owner = ServingOwner::start( + context.clone(), + q.clone(), + input(&f, 221, "owner", 1_000), + identity()?, + ) + .await?; + ready(&owner).await?; + let snapshot = owner.snapshot(Some("owner".into())).await?; + budget.close(); + owner.close(); + assert!( + ServingOwner::start( + context, + q.clone(), + input(&f, 222, "owner", DEFAULT_LEASE_MS), + identity()? + ) + .await + .is_err() + ); + assert!(matches!( + snapshot.headers(&[missing(&f)?]).await, + Err(ServingReadError::Inactive) + )); + tokio::time::sleep(Duration::from_millis(500)).await; + assert_eq!(pin_count(&f).await?, 1); + drop(snapshot); + assert_eq!( + timeout(Duration::from_secs(8), owner.close_and_drain()) + .await? + .phase, + ServingOwnerPhase::Released + ); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn shared_owner_and_snapshot_budgets_preserve_account_shares_and_physical_read_slots() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let context = shared_context(&f, store, &root, ServingReadBudget::new(4, tasks.clone())?)?; + let mut owners = Vec::new(); + for reader in 223..=224 { + owners.push( + ServingOwner::start( + context.clone(), + q.clone(), + input(&f, reader, "owner", DEFAULT_LEASE_MS), + identity()?, + ) + .await?, + ); + } + assert!( + ServingOwner::start( + context.clone(), + q.clone(), + input(&f, 225, "owner", DEFAULT_LEASE_MS), + identity()? + ) + .await + .is_err() + ); + owners.push( + ServingOwner::start( + context, + q.clone(), + input(&f, 226, "viewer", DEFAULT_LEASE_MS), + identity()?, + ) + .await?, + ); + for owner in &owners { + ready(owner).await?; + } + assert_eq!(pin_count(&f).await?, 3); + let first = owners[0].snapshot(Some("owner".into())).await?; + let second = owners[0].snapshot(Some("owner".into())).await?; + assert!(owners[0].snapshot(Some("owner".into())).await.is_err()); + let other = owners[0].snapshot(Some("viewer".into())).await?; + assert_eq!(first.headers(&[missing(&f)?]).await?, vec![None]); + assert_eq!(other.headers(&[missing(&f)?]).await?, vec![None]); + drop(first); + drop(second); + drop(other); + for owner in owners { + assert_eq!( + timeout(Duration::from_secs(8), owner.close_and_drain()) + .await? + .phase, + ServingOwnerPhase::Released + ); + } + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/lifecycle/restarts.rs b/crates/canopy-server/src/packs/publication/tests/serving/lifecycle/restarts.rs new file mode 100644 index 00000000..c77e380b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/lifecycle/restarts.rs @@ -0,0 +1,226 @@ +use super::*; + +#[tokio::test] +async fn known_registration_denial_closes_without_a_pin_or_namespace_and_returns_owner_admission() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let context = shared_context(&f, store, &root, ServingReadBudget::new(2, tasks.clone())?)?; + let before = f.counts().await?; + for reader in 229..=231 { + let owner = ServingOwner::start( + context.clone(), + q.clone(), + input(&f, reader, "other", DEFAULT_LEASE_MS), + identity()?, + ) + .await?; + let result = timeout(Duration::from_secs(8), owner.drain_observer().wait()).await?; + assert_eq!(result.phase, ServingOwnerPhase::Denied); + assert!(result.token.is_none()); + assert_eq!(pin_count(&f).await?, 0); + assert_eq!(f.counts().await?, before); + assert!(owner.snapshot(Some("other".into())).await.is_err()); + } + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn lost_acquisition_ack_then_read_revocation_still_hands_off_and_authentically_drains() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (dispatch, entered) = q.pause_for_test().await; + q.fault_for_test(2); + let owner = ServingOwner::start( + context(&f, store, &root, tasks.clone())?, + q.clone(), + input(&f, 232, "viewer", DEFAULT_LEASE_MS), + identity()?, + ) + .await?; + let (recover, recovering) = owner.pause_recovery_for_test().await; + timeout(Duration::from_secs(8), entered).await??; + dispatch.send(()).map_err(|_| "dispatch disappeared")?; + timeout(Duration::from_secs(8), recovering).await??; + let ticket = q + .pending_serving_command([232; 16]) + .await + .ok_or("owned uncertainty disappeared")?; + assert!(matches!(ticket.state(), PublicationState::Uncertain(_))); + assert_eq!(pin_count(&f).await?, 1); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + recover.send(()).map_err(|_| "recovery owner disappeared")?; + let result = timeout(Duration::from_secs(8), owner.drain_observer().wait()).await?; + assert_eq!(result.phase, ServingOwnerPhase::Released); + assert!(result.token.is_some()); + assert!(matches!( + ticket.state(), + PublicationState::Finished(Ok(PublicationOutcome::ServingCommand(_))) + )); + assert_eq!(pin_count(&f).await?, 0); + assert!(owner.snapshot(Some("viewer".into())).await.is_err()); + let saved = + RegisteredCustody::load_for(&f.client(), &f.target, CustodyPurpose::Serving, [232; 16]) + .await? + .ok_or("recorded grant lost")?; + assert!(matches!( + saved.recover_serving(&f.client()).await?.output, + ServingReply::Granted(_) + )); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn producer_restarts_preserve_factory_identity_held_ticket_capture_renewal_and_release() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for point in 1..=5 { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let first = identity()?; + let owner = ServingOwner::start( + context(&f, store, &root, tasks.clone())?, + q.clone(), + input(&f, 227, "owner", 2_000), + first, + ) + .await?; + // The current-thread runtime cannot run the spawned producer until + // this caller next yields, after its failure point is configured. + owner.fault_for_test(point); + let stats = ready(&owner).await?; + let token = stats.token.ok_or("token")?; + if point == 4 { + timeout(Duration::from_secs(8), async { + while owner.stats().renewals == 0 { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + } + if point <= 3 { + let saved = RegisteredCustody::load_for( + &f.client(), + &f.target, + CustodyPurpose::Serving, + [227; 16], + ) + .await? + .ok_or("original")?; + assert_eq!(saved.evidence().identity().request_id, first.request_id); + assert!(matches!( + f.client().resolve(saved.evidence()).await?, + Resolution::Committed(_) + )); + assert_eq!( + token.admission_sequence, + saved + .recover_serving(&f.client()) + .await? + .receipt + .commit_sequence + ); + } + let final_state = timeout(Duration::from_secs(8), owner.close_and_drain()).await?; + assert_eq!(final_state.phase, ServingOwnerPhase::Released); + assert!(final_state.retries >= 1, "point {point}: {final_state:?}"); + assert_eq!(final_state.token, Some(token)); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn owner_drain_keeps_cell_root_and_workspace_until_detached_provider_work_finishes() -> Result +{ + use std::sync::atomic::Ordering; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::super::blocked::Gate::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let owner = ServingOwner::start( + context(&f, store, &root, tasks.clone())?, + q.clone(), + input(&f, 228, "owner", DEFAULT_LEASE_MS), + identity()?, + ) + .await?; + ready(&owner).await?; + let snapshot = owner.snapshot(Some("owner".into())).await?; + provider.armed.store(true, Ordering::Release); + let oid = missing(&f)?; + let observed = tokio::spawn(async move { snapshot.headers(&[oid]).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + observed.abort(); + assert!( + observed + .await + .err() + .ok_or("provider completed")? + .is_cancelled() + ); + let closing = owner.clone(); + let drain = tokio::spawn(async move { closing.close_and_drain().await }); + tokio::time::sleep(Duration::from_millis(100)).await; + assert!(!drain.is_finished()); + assert!(!tasks.is_empty()); + assert_eq!(pin_count(&f).await?, 1); + assert!(root.path().exists()); + assert!(matches!( + owner.snapshot(Some("owner".into())).await, + Err(ServingReadError::Inactive) + )); + provider.proceed.add_permits(1); + assert_eq!( + timeout(Duration::from_secs(8), drain).await??.phase, + ServingOwnerPhase::Released + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/pool.rs b/crates/canopy-server/src/packs/publication/tests/serving/pool.rs new file mode 100644 index 00000000..bc473428 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/pool.rs @@ -0,0 +1,313 @@ +//! Bounded pooling and actual lifecycle ownership, including head races. +use super::*; + +pub(super) fn pooled( + f: &Fixture, + store: Arc, + root: &tempfile::TempDir, + tasks: TaskTracker, + q: PublicationCoordinator, +) -> Result { + Ok(ServingPool::new( + ServingContext::new( + f.client(), + f.target.clone(), + f.authority(), + Arc::new(CatalogIndexes::new(store.clone(), f.format)), + Arc::new(CatalogFiles::new( + root.path(), + DiskBudget::new(64 << 20), + store, + f.format, + CatalogFileLimits::default(), + )?), + ServingReadBudget::new(32, tasks)?, + "owner".into(), + )?, + q, + ServingPoolLimits::default(), + )?) +} +pub(super) fn queue(f: &Fixture) -> Result { + Ok(PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?) +} +async fn advance(f: &Fixture, generation: u64) -> Result { + // Copied certified roots isolate selection/lifecycle semantics; this is not + // native publication, changing Git content, or a full-history benchmark. + edit(f, &format!("INSERT INTO catalog_generations(generation,catalog,certificate,refs) SELECT {generation},catalog,certificate,refs FROM catalog_generations WHERE generation=1; UPDATE catalog_state SET generation={generation} WHERE singleton=1")).await +} +pub(super) async fn finish( + f: &Fixture, + pool: &ServingPool, + q: &PublicationCoordinator, + tasks: TaskTracker, +) -> Result { + timeout(Duration::from_secs(8), pool.close_and_drain()).await?; + assert_eq!(pin_count(f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + Ok(()) +} + +#[tokio::test] +async fn concurrent_viewers_share_one_generation_and_every_cached_borrow_checks_access() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let before = f.counts().await?; + let snapshots = + futures_util::future::join_all((0..12).map(|_| pool.snapshot(Some("viewer".into())))) + .await + .into_iter() + .collect::, _>>()?; + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(pool.owners_for_test().await.len(), 1); + assert!(snapshots.iter().all(|s| s.fact() == snapshots[0].fact())); + assert_eq!(f.counts().await?, before); + assert!(pool.snapshot(Some("other".into())).await.is_err()); + assert!(pool.snapshot(None).await.is_err()); + edit(&f, "UPDATE ref_generation SET visibility='public'").await?; + let anonymous = pool.snapshot(None).await?; + assert_eq!(anonymous.fact(), snapshots[0].fact()); + assert_eq!(pin_count(&f).await?, 1); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'; UPDATE ref_generation SET visibility='private'").await?; + assert!(pool.snapshot(Some("viewer".into())).await.is_err()); + assert!(snapshots[0].headers(&[missing(&f)?]).await.is_err()); + assert!(anonymous.headers(&[missing(&f)?]).await.is_err()); + drop((snapshots, anonymous)); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn canceled_cold_observer_and_lost_ack_keep_one_owned_acquisition() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let (dispatch, entered) = q.pause_for_test().await; + q.fault_for_test(2); + let work = pool.clone(); + let observer = tokio::spawn(async move { work.snapshot(Some("owner".into())).await }); + timeout(Duration::from_secs(8), entered).await??; + observer.abort(); + assert!( + observer + .await + .err() + .ok_or("completed early")? + .is_cancelled() + ); + dispatch.send(()).map_err(|_| "dispatch gone")?; + let snapshot = + timeout(Duration::from_secs(8), pool.snapshot(Some("owner".into()))).await??; + assert_eq!(snapshot.fact().generation, 1); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(pool.owners_for_test().await.len(), 1); + drop(snapshot); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn four_generation_bound_retains_borrows_and_reuses_only_actually_drained_slots() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let mut snapshots = Vec::new(); + for generation in 1..=4 { + if generation > 1 { + advance(&f, generation).await?; + } + let snapshot = pool.snapshot(Some("owner".into())).await?; + assert_eq!(snapshot.fact().generation, generation); + snapshots.push(snapshot); + } + advance(&f, 5).await?; + assert!(matches!( + pool.snapshot(Some("owner".into())).await, + Err(ServingOwnerError::Read(ServingReadError::Capability( + Error::Capacity("repository serving generations") + ))) + )); + assert_eq!(pin_count(&f).await?, 4); + let old = pool.owners_for_test().await.remove(0); + drop(snapshots.remove(0)); + assert!(pool.snapshot(Some("owner".into())).await.is_err()); + assert_eq!( + timeout(Duration::from_secs(8), old.drain_observer().wait()) + .await? + .phase, + ServingOwnerPhase::Released + ); + let fifth = pool.snapshot(Some("owner".into())).await?; + assert_eq!(fifth.fact().generation, 5); + assert_eq!(pin_count(&f).await?, 4); + for snapshot in &snapshots { + assert_eq!(snapshot.headers(&[missing(&f)?]).await?, vec![None]); + } + drop((snapshots, fifth)); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn acquisition_head_race_returns_actual_accepted_fact_and_reuses_it() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let (dispatch, entered) = q.pause_for_test().await; + let work = pool.clone(); + let observer = tokio::spawn(async move { work.snapshot(Some("owner".into())).await }); + timeout(Duration::from_secs(8), entered).await??; + advance(&f, 2).await?; + dispatch.send(()).map_err(|_| "dispatch gone")?; + let snapshot = timeout(Duration::from_secs(8), observer).await???; + assert_eq!(snapshot.fact().generation, 2); + let second = pool.snapshot(Some("owner".into())).await?; + assert_eq!(second.fact(), snapshot.fact()); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(pool.owners_for_test().await.len(), 1); + drop((snapshot, second)); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn busy_eviction_resumes_and_canceled_exclusive_drain_joins_exact_uncertain_release() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let snapshot = pool.snapshot(Some("owner".into())).await?; + assert!(!pool.quiesce().await?); + assert_eq!(snapshot.headers(&[missing(&f)?]).await?, vec![None]); + drop(snapshot); + let held = q.try_reserve( + ReadyServingCommand::acquire( + f.client(), + f.target.clone(), + f.begin([233; 16]), + identity()?, + f.authority(), + ) + .await?, + )?; + assert!(!pool.quiesce().await?); + assert!(!q.stats().await.closed); + drop(pool.snapshot(Some("owner".into())).await?); + held.discard_held().await?; + let (dispatch, entered) = q.pause_for_test().await; + q.fault_for_test(2); + let work = pool.clone(); + let observed = tokio::spawn(async move { work.quiesce().await }); + timeout(Duration::from_secs(8), entered).await??; + observed.abort(); + assert!(observed.await.err().ok_or("drained early")?.is_cancelled()); + assert_eq!(pin_count(&f).await?, 1); + dispatch.send(()).map_err(|_| "dispatch gone")?; + timeout(Duration::from_secs(8), pool.close_and_drain()).await?; + assert_eq!(pin_count(&f).await?, 0); + assert!(q.stats().await.closed); + assert!(pool.quiesce().await?); + assert!(pool.snapshot(Some("owner".into())).await.is_err()); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn blocked_old_generation_does_not_block_other_release_or_allow_early_eviction() -> Result { + use std::sync::atomic::Ordering; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let snapshot = pool.snapshot(Some("owner".into())).await?; + provider.armed.store(true, Ordering::Release); + let oid = missing(&f)?; + let observer = tokio::spawn(async move { snapshot.headers(&[oid]).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + observer.abort(); + assert!( + observer + .await + .err() + .ok_or("provider finished")? + .is_cancelled() + ); + assert!(!timeout(Duration::from_secs(1), pool.quiesce()).await??); + advance(&f, 2).await?; + let other = pool.snapshot(Some("owner".into())).await?; + let owners = pool.owners_for_test().await; + let first = owners[0].drain_observer(); + let second = owners[1].drain_observer(); + pool.close(); + drop(other); + assert_eq!( + timeout(Duration::from_secs(8), second.wait()).await?.phase, + ServingOwnerPhase::Released + ); + assert_eq!(pin_count(&f).await?, 1); + assert!( + timeout(Duration::from_millis(50), first.wait()) + .await + .is_err() + ); + assert!(root.path().exists()); + provider.proceed.add_permits(1); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/refs.rs b/crates/canopy-server/src/packs/publication/tests/serving/refs.rs new file mode 100644 index 00000000..f083aaa9 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/refs.rs @@ -0,0 +1,357 @@ +//! Immutable joint refs, bounded pagination and real retained provider work. +use super::pool::{finish, pooled, queue}; +use super::*; +use crate::RefExpectation; +use crate::packs::ref_state::{ + RefStateRecord, RefStateSnapshot, RefStateSnapshotRoot, RefStateTree, +}; + +fn operation(sequence: u64) -> [u8; 16] { + let mut operation = *b"CANOPY0100000000"; + operation[8..].copy_from_slice(&sequence.to_be_bytes()); + operation +} + +async fn install( + f: &Fixture, + store: &ArtifactStore, + base: GenerationFact, + generation: u64, + snapshot: RefStateSnapshot, +) -> Result { + // Trusted root injection isolates reader behavior; it is not native graph + // closure or evidence that the production publisher accepts these tips. + let root = RefStateSnapshotRoot::upload(store, operation(1_000 + generation), snapshot).await?; + f.install_generation( + generation, + base.catalog.ok_or("catalog absent")?, + Some(root), + ) + .await +} +fn snapshot(f: &Fixture, generation: u64) -> RefStateSnapshot { + RefStateSnapshot { + repository: f.repository, + format: f.format, + generation, + default_branch: "refs/heads/main".into(), + root: None, + } +} + +#[tokio::test] +async fn immutable_ref_pages_preserve_versions_skip_deleted_subtrees_and_recheck_cached_access() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let base = initialize(&f, store.clone()).await?; + let oid = missing(&f)?; + let mut records = Vec::new(); + for n in 0..512 { + records.push(RefStateRecord::new( + &format!("refs/heads/deleted/{n:04}"), + RefExpectation { + oid: None, + version: 2, + }, + format, + )?); + } + for n in 0..300 { + records.push(RefStateRecord::new( + &format!("refs/heads/live/{n:04}"), + RefExpectation { + oid: Some(oid), + version: 1, + }, + format, + )?); + } + let tree = RefStateTree::new(store.clone(), format); + let mut old = snapshot(&f, 1); + old.default_branch = "refs/heads/live/0000".into(); + old.root = tree + .build_sorted(operation(141), records.into_iter().map(Ok)) + .await?; + install(&f, &store, base, 2, old).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store.clone(), &root, tasks.clone(), q.clone())?; + let view = pool.snapshot(Some("viewer".into())).await?; + let resolved = view.resolve_ref(None).await?; + assert_eq!(resolved.generation, 1); + assert_eq!(resolved.reference, "refs/heads/live/0000"); + assert_eq!( + resolved.state, + Some(RefExpectation { + oid: Some(oid), + version: 1 + }) + ); + assert!( + view.resolve_ref(Some("refs/heads/absent")) + .await? + .state + .is_none() + ); + assert_eq!( + view.resolve_ref(Some("refs/heads/deleted/0000")) + .await? + .state, + Some(RefExpectation { + oid: None, + version: 2 + }) + ); + let first = view.refs_page("", None, true).await?; + assert_eq!(first.refs.len(), 256); + assert!(first.has_more); + assert_eq!(first.refs[0].0, "refs/heads/live/0000"); + let after = &first.refs.last().ok_or("empty first page")?.0; + let last = view.refs_page(after, Some(first.generation), true).await?; + assert_eq!(last.refs.len(), 44); + assert!(!last.has_more); + assert_eq!( + last.refs.last().ok_or("empty last page")?.0, + "refs/heads/live/0299" + ); + let all = view.refs_page("", None, false).await?; + assert_eq!(all.refs.len(), 256); + assert!(all.refs.iter().all(|(_, state)| state.oid.is_none())); + assert!(matches!( + view.refs_page(after, Some(0), true).await, + Err(ServingReadError::Changed) + )); + assert!(view.refs_page(after, None, true).await.is_err()); + assert!( + view.resolve_ref(Some("refs/heads/bad..name")) + .await + .is_err() + ); + let mut new = snapshot(&f, 2); + new.default_branch = "refs/heads/new".into(); + install(&f, &store, base, 3, new).await?; + let current = pool.snapshot(Some("owner".into())).await?; + assert_eq!(current.resolve_ref(None).await?.reference, "refs/heads/new"); + assert_eq!( + view.resolve_ref(None).await?.reference, + "refs/heads/live/0000" + ); + assert!(matches!( + current.refs_page(after, Some(1), true).await, + Err(ServingReadError::Changed) + )); + edit(&f, "UPDATE ref_generation SET visibility='public'").await?; + let public = pool.snapshot(None).await?; + assert_eq!(public.resolve_ref(None).await?.generation, 2); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'; UPDATE ref_generation SET visibility='private'").await?; + assert!(view.resolve_ref(None).await.is_err()); + assert!(view.refs_page("", None, true).await.is_err()); + assert!(public.resolve_ref(None).await.is_err()); + drop((view, current, public)); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn ref_page_byte_bound_continues_exactly_after_long_names_without_losing_entries() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let base = initialize(&f, store.clone()).await?; + let oid = missing(&f)?; + let names: Vec<_> = (0..9) + .map(|n| format!("refs/heads/{n:02}-{}", "x".repeat(65_000))) + .collect(); + let records = names + .iter() + .map(|name| { + RefStateRecord::new( + name, + RefExpectation { + oid: Some(oid), + version: 1, + }, + format, + ) + }) + .collect::, _>>()?; + let mut refs = snapshot(&f, 1); + refs.root = RefStateTree::new(store.clone(), format) + .build_sorted(operation(142), records.into_iter().map(Ok)) + .await?; + install(&f, &store, base, 2, refs).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let view = pool.snapshot(Some("owner".into())).await?; + let first = view.refs_page("", None, true).await?; + assert_eq!(first.refs.len(), 8); + assert!(first.has_more); + assert!( + first + .refs + .iter() + .map(|(name, _)| name.len() + 64) + .sum::() + <= 512 * 1024 + ); + let after = &first.refs.last().ok_or("first page")?.0; + let last = view.refs_page(after, Some(first.generation), true).await?; + assert_eq!(last.refs.len(), 1); + assert!(!last.has_more); + assert_eq!( + first + .refs + .into_iter() + .chain(last.refs) + .map(|(name, _)| name) + .collect::>(), + names + ); + drop(view); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn blocked_ref_download_survives_canceled_observation_and_refuses_early_pin_release() -> Result +{ + use std::sync::atomic::Ordering; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let view = pool.snapshot(Some("owner".into())).await?; + provider.armed.store(true, Ordering::Release); + let observer = tokio::spawn(async move { view.resolve_ref(None).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + observer.abort(); + assert!( + observer + .await + .err() + .ok_or("ref observer finished")? + .is_cancelled() + ); + assert!(!pool.quiesce().await?); + let mut drain = tokio::spawn({ + let pool = pool.clone(); + async move { pool.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + provider.proceed.add_permits(1); + timeout(Duration::from_secs(8), drain).await??; + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn ref_download_drains_but_never_returns_a_result_after_viewer_revocation() -> Result { + use std::sync::atomic::Ordering; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), f.repository)); + initialize(&f, store.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store, &root, tasks.clone(), q.clone())?; + let view = pool.snapshot(Some("viewer".into())).await?; + provider.armed.store(true, Ordering::Release); + let observer = tokio::spawn(async move { view.resolve_ref(None).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + provider.proceed.add_permits(1); + assert!(matches!( + timeout(Duration::from_secs(8), observer).await??, + Err(ServingReadError::Inactive) + )); + let authorized = pool.snapshot(Some("owner".into())).await?; + assert_eq!( + authorized.resolve_ref(None).await?.reference, + "refs/heads/main" + ); + assert!(pool.snapshot(Some("viewer".into())).await.is_err()); + drop(authorized); + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn ref_snapshot_context_mismatch_never_falls_back_to_legacy_sql() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let base = initialize(&f, store.clone()).await?; + let q = queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let pool = pooled(&f, store.clone(), &root, tasks.clone(), q.clone())?; + for (generation, other_format, ref_generation) in [ + (2, format, 3), + ( + 3, + if format == ObjectFormat::Sha1 { + ObjectFormat::Sha256 + } else { + ObjectFormat::Sha1 + }, + 1, + ), + ] { + let mut bad = snapshot(&f, ref_generation); + bad.format = other_format; + install(&f, &store, base, generation, bad).await?; + let view = pool.snapshot(Some("owner".into())).await?; + assert!(matches!( + view.resolve_ref(None).await, + Err(ServingReadError::Context) + )); + assert!(matches!( + view.refs_page("", None, true).await, + Err(ServingReadError::Context) + )); + drop(view); + } + finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/selection_drain.rs b/crates/canopy-server/src/packs/publication/tests/serving/selection_drain.rs new file mode 100644 index 00000000..d382da6d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/selection_drain.rs @@ -0,0 +1,624 @@ +//! Indexed current-root observation and exact eviction admission, with real receipts. +use super::*; +use cellule_runtime::Resolution; + +async fn selection(f: &Fixture, actor: Option<&str>) -> Result> { + Ok(f.client() + .query::( + &f.target, + None, + ServingSelection { + repository: f.repository, + actor: actor.map(str::to_owned), + }, + ) + .await? + .output) +} +fn queue(f: &Fixture) -> Result { + Ok(PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?) +} +async fn pin( + f: &Fixture, + store: Arc, + root: &tempfile::TempDir, + tasks: TaskTracker, + reader: u8, +) -> Result { + let (lease, _, _) = acquire(f, Some("owner"), reader, DEFAULT_LEASE_MS).await?; + Ok(ServingPin::open( + context(f, store, root, tasks)?, + lease.token, + Some("owner".into()), + ) + .await?) +} +async fn acquisition(f: &Fixture, operation: u8) -> Result { + Ok(ReadyServingCommand::acquire( + f.client(), + f.target.clone(), + f.begin([operation; 16]), + identity()?, + f.authority(), + ) + .await?) +} +async fn close_drained(gate: &ServingDrainAdmission) -> Result { + timeout(Duration::from_secs(5), async { + while !gate.close_if_drained().await { + tokio::task::yield_now().await; + } + }) + .await?; + Ok(()) +} +async fn release_result(ticket: &PublicationTicket) -> Result> { + match timeout(Duration::from_secs(5), ticket.wait()).await? { + PublicationState::Finished(Ok(PublicationOutcome::ServingRelease(value))) => Ok(value), + state => Err(format!("release state {state:?}").into()), + } +} + +#[tokio::test] +async fn selection_requires_current_read_joint_initialization_and_exact_repository_identity() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + assert!(selection(&f, Some("owner")).await?.is_none()); + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let fact = initialize(&f, store).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let before = f.counts().await?; + assert_eq!(selection(&f, Some("viewer")).await?, Some(fact)); + assert_eq!(selection(&f, Some("owner")).await?, Some(fact)); + assert!(selection(&f, Some("other")).await?.is_none()); + assert!(selection(&f, None).await?.is_none()); + let wrong = ServingSelection { + repository: *uuid::Uuid::new_v4().as_bytes(), + actor: Some("owner".into()), + }; + assert!( + f.client() + .query::(&f.target, None, wrong) + .await? + .output + .is_none() + ); + edit(&f, "UPDATE ref_generation SET visibility='public'").await?; + assert_eq!(selection(&f, None).await?, Some(fact)); + edit(&f, "UPDATE ref_generation SET visibility='private'; DELETE FROM repository_members WHERE account='viewer'").await?; + assert!(selection(&f, None).await?.is_none()); + assert!(selection(&f, Some("viewer")).await?.is_none()); + assert_eq!(f.counts().await?, before); + assert_eq!(pin_count(&f).await?, 0); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn selection_follows_current_joint_head_while_old_pin_remains_immutable() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let initial = initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let old = pin(&f, store, &root, tasks.clone(), 201).await?; + // Trusted copied roots qualify head selection, not native publication. + edit(&f, "INSERT INTO catalog_generations(generation,catalog,certificate,refs) SELECT 2,catalog,certificate,refs FROM catalog_generations WHERE generation=1; UPDATE catalog_state SET generation=2 WHERE singleton=1").await?; + let current = selection(&f, Some("owner")) + .await? + .ok_or("current root missing")?; + assert_eq!(current.generation, 2); + assert_eq!( + (current.catalog, current.refs, current.certificate), + (initial.catalog, initial.refs, initial.certificate) + ); + assert_eq!(old.fact(), initial); + assert_eq!( + f.client() + .query::( + &f.target, + None, + ServingCheck { + token: old.token(), + actor: Some("owner".into()), + } + ) + .await? + .output + .ok_or("old retention lost")? + .fact, + initial + ); + f.handle.query(0,4096,|db| { + let mut query=db.prepare("EXPLAIN QUERY PLAN SELECT g.generation,g.catalog,g.certificate,g.refs FROM catalog_state s JOIN catalog_generations g ON g.generation=s.generation WHERE s.singleton=1")?; + let plan:Vec=query.query_map([],|r|r.get(3))?.collect::>()?; + assert!(plan.iter().all(|line| !line.contains("SCAN") && !line.contains("TEMP B-TREE")),"{plan:?}"); + Ok(Vec::new()) + }).await?; + release(&f, &old).await?; + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[test] +fn selection_codec_is_bounded_and_does_not_carry_a_pin() -> Result { + let repository = *uuid::Uuid::new_v4().as_bytes(); + for actor in [None, Some("viewer".into())] { + let input = ServingSelection { repository, actor }; + let mut encoder = BoundedEncoder::new(1024)?; + input.encode(&mut encoder)?; + let bytes = encoder.finish(); + let mut decoder = BoundedDecoder::new(&bytes, 1024)?; + assert_eq!(ServingSelection::decode(&mut decoder)?, input); + decoder.finish()?; + for end in 0..bytes.len() { + assert!( + (|| -> std::result::Result<(), CodecError> { + let mut decoder = BoundedDecoder::new(&bytes[..end], 1024)?; + ServingSelection::decode(&mut decoder)?; + decoder.finish() + })() + .is_err() + ); + } + let mut trailing = bytes; + trailing.push(0); + let mut decoder = BoundedDecoder::new(&trailing, 1024)?; + ServingSelection::decode(&mut decoder)?; + assert!(decoder.finish().is_err()); + } + assert!( + ServingSelection { + repository: [0; 16], + actor: None + } + .encode(&mut BoundedEncoder::new(1024)?) + .is_err() + ); + Ok(()) +} + +#[tokio::test] +async fn eviction_admits_only_exact_selected_releases_and_closes_after_all_real_successes() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let first = pin(&f, store.clone(), &root, tasks.clone(), 202).await?; + let second = pin(&f, store.clone(), &root, tasks.clone(), 203).await?; + let foreign = pin(&f, store, &root, tasks.clone(), 204).await?; + let q = queue(&f)?; + let gate = q + .reserve_serving_drain(&[first.token(), second.token()]) + .await? + .ok_or("idle drain refused")?; + assert!(!gate.close_if_drained().await); + assert!(!q.close_if_idle().await); + assert!(q.reserve_serving_drain(&[]).await?.is_none()); + let before = f.counts().await?; + let denied = q + .try_reserve(acquisition(&f, 202).await?) + .err() + .ok_or("acquisition admitted during drain")?; + assert_eq!(denied.reason, PublicationScheduleError::Closed); + assert_eq!(f.counts().await?, before); + let denied_release = q + .try_reserve(foreign.ready_release(identity()?).await?) + .err() + .ok_or("foreign release admitted during drain")?; + assert_eq!(denied_release.reason, PublicationScheduleError::Closed); + assert_eq!(pin_count(&f).await?, 3); + let held = q.try_reserve(first.ready_release(identity()?).await?)?; + assert!(!gate.close_if_drained().await); + held.activate().await?; + assert_eq!( + release_result(&held).await?.output, + ServingReleaseReply::Released + ); + assert!(!gate.close_if_drained().await); + let held = q.try_reserve(second.ready_release(identity()?).await?)?; + held.activate().await?; + assert_eq!( + release_result(&held).await?.output, + ServingReleaseReply::Released + ); + close_drained(&gate).await?; + drop(gate); + assert!(q.stats().await.closed); + assert_eq!( + q.try_reserve(denied.ready) + .err() + .ok_or("closed admission reopened")? + .reason, + PublicationScheduleError::Closed + ); + assert!(q.close_and_drain().await.is_empty()); + release(&f, &foreign).await?; + assert_eq!(pin_count(&f).await?, 0); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn busy_or_invalid_drain_never_changes_existing_admission_or_exact_evidence() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let q = queue(&f)?; + let held = q.try_reserve(acquisition(&f, 205).await?)?; + assert!(q.reserve_serving_drain(&[]).await?.is_none()); + assert!(!q.stats().await.closed); + held.discard_held().await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let retained = pin(&f, store.clone(), &root, tasks.clone(), 206).await?; + for tokens in [vec![retained.token(); 2], vec![retained.token(); 17]] { + assert!(matches!( + q.reserve_serving_drain(&tokens).await, + Err(PublicationScheduleError::InvalidLimits) + )); + } + let mut malformed = retained.token(); + malformed.admission_sequence = 0; + assert!(matches!( + q.reserve_serving_drain(&[malformed]).await, + Err(PublicationScheduleError::InvalidLimits) + )); + let mut foreign = retained.token(); + foreign.repository = *uuid::Uuid::new_v4().as_bytes(); + assert!(matches!( + q.reserve_serving_drain(&[foreign]).await, + Err(PublicationScheduleError::Foreign) + )); + let ordinary = acquisition(&f, 207).await?; + let original = ordinary.evidence().clone(); + let gate = q + .reserve_serving_drain(&[]) + .await? + .ok_or("idle admission")?; + let failure = q + .try_reserve(ordinary) + .err() + .ok_or("acquisition admitted during drain")?; + assert_eq!(failure.reason, PublicationScheduleError::Closed); + assert!(matches!( + f.client().resolve(&original).await?, + Resolution::Absent + )); + drop(gate); + let ticket = q.try_reserve(failure.ready)?; + ticket.activate().await?; + let state = timeout(Duration::from_secs(5), ticket.wait()).await?; + let PublicationState::Finished(Ok(PublicationOutcome::ServingCommand(value))) = state else { + return Err(format!("acquisition state {state:?}").into()); + }; + let lease = granted(value.output)?; + let resumed = ServingPin::open( + context(&f, store, &root, tasks.clone())?, + lease.token, + Some("owner".into()), + ) + .await?; + assert!(matches!( + f.client().resolve(&original).await?, + Resolution::Committed(_) + )); + assert!(!q.stats().await.closed); + release(&f, &retained).await?; + release(&f, &resumed).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn release_uncertainty_and_observer_loss_cannot_complete_eviction_early() -> Result { + for fault in 1..=3 { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let retained = pin(&f, store, &root, tasks.clone(), 208).await?; + let q = queue(&f)?; + let gate = q + .reserve_serving_drain(&[retained.token()]) + .await? + .ok_or("idle drain")?; + let original = retained.ready_release(identity()?).await?; + let evidence = original.evidence().clone(); + q.fault_for_test(fault); + let ticket = q.submit(original).await?; + assert!(matches!( + timeout(Duration::from_secs(5), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + assert!(!gate.close_if_drained().await); + assert_eq!(q.stats().await.command_bytes, 8 << 10); + drop(ticket); + let ticket = q + .pending_serving_release([208; 16]) + .await + .ok_or("release owner lost")?; + ticket.recover().await?; + assert_eq!( + release_result(&ticket).await?.output, + ServingReleaseReply::Released + ); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Committed(_) + )); + close_drained(&gate).await?; + assert_eq!(q.stats().await.command_bytes, 0); + drop(gate); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn a_denied_release_keeps_its_sql_root_and_cannot_satisfy_the_drain_guard() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let retained = pin(&f, store, &root, tasks.clone(), 209).await?; + let q = queue(&f)?; + let gate = q + .reserve_serving_drain(&[retained.token()]) + .await? + .ok_or("idle drain")?; + let ready = retained.ready_release(identity()?).await?; + edit(&f, "UPDATE repository_identity SET owner='other'").await?; + let state = timeout(Duration::from_secs(5), q.submit(ready).await?.wait()).await?; + assert!( + matches!(state,PublicationState::Finished(Err(ref error)) if matches!(&**error, + PublicationError::ServingRelease(InvocationError::Rejected(value)) if value.output==ServingReleaseReply::Denied(ServingDenial::Unauthorized))) + ); + assert_eq!(pin_count(&f).await?, 1); + assert!(!gate.close_if_drained().await); + assert!(!q.close_if_idle().await); + drop(gate); + edit(&f, "UPDATE repository_identity SET owner='owner'").await?; + release(&f, &retained).await?; + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn global_close_waits_for_drain_owner_and_cancellation_never_reopens_closed_admission() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let q = queue(&f)?; + let gate = q.reserve_serving_drain(&[]).await?.ok_or("idle drain")?; + let closing = q.clone(); + let waiter = tokio::spawn(async move { closing.close_and_drain().await }); + tokio::task::yield_now().await; + assert!(!waiter.is_finished()); + waiter.abort(); + assert!( + waiter + .await + .err() + .ok_or("global close completed before drain")? + .is_cancelled() + ); + assert!(!q.stats().await.closed); + assert!(gate.close_if_drained().await); + drop(gate); + assert!(q.stats().await.closed); + assert!(q.close_and_drain().await.is_empty()); + assert!(q.reserve_serving_drain(&[]).await?.is_none()); + + let q = queue(&f)?; + let gate = q + .reserve_serving_drain(&[]) + .await? + .ok_or("new idle drain")?; + let closing = q.clone(); + let waiter = tokio::spawn(async move { closing.close_and_drain().await }); + tokio::task::yield_now().await; + assert!(!waiter.is_finished()); + drop(gate); + assert!(timeout(Duration::from_secs(5), waiter).await??.is_empty()); + assert!(q.stats().await.closed); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn reused_reader_id_cannot_release_another_exact_drain_token() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let old = pin(&f, store.clone(), &root, tasks.clone(), 210).await?; + let old_token = old.token(); + release(&f, &old).await?; + let current = pin(&f, store, &root, tasks.clone(), 210).await?; + assert_eq!(old_token.reader, current.token().reader); + assert_ne!( + old_token.admission_sequence, + current.token().admission_sequence + ); + let q = queue(&f)?; + let gate = q + .reserve_serving_drain(&[old_token]) + .await? + .ok_or("idle drain")?; + let ready = current.ready_release(identity()?).await?; + let original = ready.evidence().clone(); + let refused = q + .try_reserve(ready) + .err() + .ok_or("different token admitted")?; + assert_eq!(refused.reason, PublicationScheduleError::Closed); + assert!(matches!( + f.client().resolve(&original).await?, + Resolution::Absent + )); + assert_eq!(pin_count(&f).await?, 1); + assert!(!gate.close_if_drained().await); + assert_eq!(q.stats().await.command_bytes, 0); + drop(gate); + let ticket = q.submit(refused.ready).await?; + assert_eq!( + release_result(&ticket).await?.output, + ServingReleaseReply::Released + ); + assert!(matches!( + f.client().resolve(&original).await?, + Resolution::Committed(_) + )); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn global_close_preserves_release_admission_through_real_dispatch_and_uncertainty() -> Result +{ + for fault in 1..=3 { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let retained = pin(&f, store, &root, tasks.clone(), 211).await?; + let q = queue(&f)?; + let gate = q + .reserve_serving_drain(&[retained.token()]) + .await? + .ok_or("idle drain")?; + let closing = q.clone(); + let waiter = tokio::spawn(async move { closing.close_and_drain().await }); + tokio::task::yield_now().await; + assert!(!waiter.is_finished()); + assert!(!q.stats().await.closed); + let (dispatch, entered) = q.pause_for_test().await; + q.fault_for_test(fault); + let ticket = q.submit(retained.ready_release(identity()?).await?).await?; + timeout(Duration::from_secs(5), entered).await??; + assert!(!waiter.is_finished()); + assert!(!gate.close_if_drained().await); + assert_eq!(pin_count(&f).await?, 1); + dispatch + .send(()) + .map_err(|_| "release worker disappeared")?; + assert!(matches!( + timeout(Duration::from_secs(5), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + assert!(!waiter.is_finished()); + assert!(!gate.close_if_drained().await); + ticket.recover().await?; + assert_eq!( + release_result(&ticket).await?.output, + ServingReleaseReply::Released + ); + assert!(!waiter.is_finished()); + close_drained(&gate).await?; + assert!(timeout(Duration::from_secs(5), waiter).await??.is_empty()); + drop(gate); + assert!(q.stats().await.closed); + assert_eq!(pin_count(&f).await?, 0); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn dropping_admission_guard_preserves_dispatched_original_and_its_recovery_credits() -> Result +{ + let f = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + initialize(&f, store.clone()).await?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let retained = pin(&f, store, &root, tasks.clone(), 212).await?; + let q = queue(&f)?; + let gate = q + .reserve_serving_drain(&[retained.token()]) + .await? + .ok_or("idle drain")?; + let original = retained.ready_release(identity()?).await?; + let evidence = original.evidence().clone(); + let (dispatch, entered) = q.pause_for_test().await; + q.fault_for_test(3); + let ticket = q.submit(original).await?; + timeout(Duration::from_secs(5), entered).await??; + drop(gate); + drop(ticket); + assert_eq!(q.stats().await.command_bytes, 8 << 10); + assert_eq!(pin_count(&f).await?, 1); + assert!(!q.stats().await.closed); + // Guard cancellation resumes admission; it cannot cancel admitted work. + let held = q.try_reserve(acquisition(&f, 213).await?)?; + assert_eq!(q.stats().await.command_bytes, (8 + 28) << 10); + dispatch + .send(()) + .map_err(|_| "release worker disappeared")?; + let ticket = q + .pending_serving_release([212; 16]) + .await + .ok_or("original lost")?; + assert!(matches!( + timeout(Duration::from_secs(5), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + held.discard_held().await?; + assert_eq!(q.stats().await.command_bytes, 8 << 10); + ticket.recover().await?; + assert_eq!( + release_result(&ticket).await?.output, + ServingReleaseReply::Released + ); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Committed(_) + )); + assert_eq!(pin_count(&f).await?, 0); + assert_eq!(q.stats().await.command_bytes, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/serving/workspace.rs b/crates/canopy-server/src/packs/publication/tests/serving/workspace.rs new file mode 100644 index 00000000..d0fe9b7d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/serving/workspace.rs @@ -0,0 +1,738 @@ +//! Native forward closures use physically verified packs. Trusted generation +//! installation isolates serving from the still-incomplete live publisher. +use super::*; + +#[tokio::test] +async fn native_write_base_streams_catalog_inputs_and_isolates_new_native_outputs() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (native, _) = super::body::catalog(&f, Arc::new(InMemory::new())).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("owner".into())).await?; + let backend = view.native_base().await?; + assert!(backend.cache.pack_sources().await?.is_empty()); + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 2); + let mut objects = crate::git_objects::GitObjects::batch_owned( + &backend.git_dir(), + &backend.cache.native, + backend.cache.clone(), + )?; + for (id, (expected, _)) in &native.fixture.objects { + let body = objects.read_verified(*expected, 1 << 20).await?; + assert_eq!(crate::object_id(format, expected.kind, &body), *id); + } + objects.finish().await?; + let path = backend.git_dir(); + assert_eq!(std::fs::read_dir(path.join("objects/pack"))?.count(), 0); + assert!(path.join("objects/info/alternates").exists()); + let body = b"new generated native object"; + let written = crate::packs::metadata::tests::git( + &path, + &["hash-object", "-w", "--stdin"], + Some(body.to_vec()), + ) + .await?; + let id = crate::object_id(format, ObjectKind::Blob, body); + assert_eq!(String::from_utf8(written)?.trim(), hex::encode(id)); + let input = format!("{}\n", hex::encode(id)); + let pack = crate::packs::metadata::tests::git( + &path, + &["pack-objects", "--stdout"], + Some(input.into_bytes()), + ) + .await?; + crate::packs::metadata::tests::git(&path, &["index-pack", "--stdin"], Some(pack)).await?; + let incoming = backend.cache.pack_sources().await?; + assert_eq!(incoming.len(), 1); + assert_eq!(incoming[0].3, [id]); // baseline history must never be ingested as the new pack + drop(view); + let mut drain = tokio::spawn({ + let pool = pool.clone(); + async move { pool.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + drop(backend); + timeout(Duration::from_secs(8), drain).await??; + super::pool::finish(&f, &pool, &q, tasks).await?; + assert!(!path.exists()); + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 0); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn native_write_base_cancellation_and_revocation_retain_real_provider_work_until_drain() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for cancel in [false, true] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = super::body::catalog(&f, provider.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("viewer".into())).await?; + assert!( + view.headers(&[*native.fixture.objects.keys().next().ok_or("object")?]) + .await?[0] + .is_some() + ); + // Warm immutable ref-root metadata so the gate below suspends the + // native pack transfer after file admission, not ref-root loading. + view.resolve_ref(None).await?; + provider.armed.store(true, Ordering::Release); + let observer = tokio::spawn(async move { view.native_base().await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + if cancel { + observer.abort(); + assert!(observer.await.err().ok_or("cancelled")?.is_cancelled()); + let mut drain = tokio::spawn({ + let pool = pool.clone(); + async move { pool.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 1); + provider.proceed.add_permits(1); + timeout(Duration::from_secs(8), drain).await??; + } else { + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + assert_eq!(pin_count(&f).await?, 1); + provider.proceed.add_permits(1); + assert!(matches!( + timeout(Duration::from_secs(8), observer).await??, + Err(ServingReadError::Inactive) + )); + } + super::pool::finish(&f, &pool, &q, tasks).await?; + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 0); + f.runtime.shutdown().await?; + } + } + Ok(()) +} +use crate::packs::catalog::serving_fixture::{operation, prepare}; +use crate::{ObjectId, ObjectKind}; +use std::collections::BTreeSet; +use std::sync::atomic::Ordering; + +#[tokio::test] +async fn complete_native_history_crosses_shards_and_wide_parent_pages_without_loose_copies() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(InMemory::new()); + // Wide fixtures exercise metadata shards, large trees and >512 parents. + let fixture = prepare(format, provider.clone(), f.repository, 600) + .await + .map_err(|e| e.to_string())?; + let store = Arc::new(ArtifactStore::new(provider, f.repository)); + initialize(&f, store.clone()).await?; + f.install_generation(2, fixture.catalog, Some(fixture.refs)) + .await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = super::body::serving_context(&f, store, &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("owner".into())).await?; + let wide = fixture.wide.ok_or("wide merge")?; + let workspace = timeout( + Duration::from_secs(60), + view.workspace(&[wide], WorkspaceLimits::default()), + ) + .await??; + let mut pending = vec![wide]; + let mut expected = BTreeSet::new(); + while let Some(id) = pending.pop() { + if expected.insert(id) { + pending.extend(fixture.edges[&id].iter().map(|e| e.child)); + } + } + assert_eq!(workspace.stats().objects, expected.len() as u64); + assert_eq!(workspace.stats().packs, 1); + for page in expected.into_iter().collect::>().chunks(512) { + assert!( + workspace + .contains(page) + .await? + .iter() + .all(|present| *present) + ); + } + assert_eq!( + workspace + .contains(&[fixture.tag, fixture.main, missing(&f)?]) + .await?, + [false, false, false] + ); + assert!(workspace.body(fixture.tag, 1024).await?.is_none()); + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 1); + let path = workspace.git_dir(); + assert_eq!(std::fs::read_dir(path.join("objects/pack"))?.count(), 2); + assert_eq!(std::fs::read_dir(path.join("objects"))?.count(), 2); // info + pack, no loose shards + // A native walk needs all ordered parents, not just the tip pack object. + let commits = crate::packs::metadata::tests::git( + &path, + &["rev-list", "--count", &hex::encode(wide)], + None, + ) + .await?; + assert_eq!(String::from_utf8(commits)?.trim(), "533"); + let body = workspace.body(wide, 64 << 10).await?.ok_or("wide body")?; + assert_eq!(crate::object_id(format, ObjectKind::Commit, &body), wide); + drop(view); + let mut drain = tokio::spawn({ + let pool = pool.clone(); + async move { pool.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + drop(workspace); + timeout(Duration::from_secs(8), drain).await??; + super::pool::finish(&f, &pool, &q, tasks).await?; + assert!(!path.exists()); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn closure_reads_reject_unselected_physical_objects_limits_and_revoked_access() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (native, _) = super::body::catalog(&f, Arc::new(InMemory::new())).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, _) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("viewer".into())).await?; + let (id, (expected, _)) = native + .fixture + .objects + .iter() + .find(|(_, (o, _))| o.kind == ObjectKind::Blob) + .ok_or("blob")?; + let workspace = view.workspace(&[*id], WorkspaceLimits::default()).await?; + assert_eq!(workspace.stats().objects, 1); + let other = *native + .fixture + .objects + .keys() + .find(|oid| *oid != id) + .ok_or("other")?; + assert_eq!( + workspace.contains(&[*id, other, *id, missing(&f)?]).await?, + [true, false, true, false] + ); + assert_eq!(workspace.body(other, 1 << 20).await?, None); + assert_eq!( + workspace.body(*id, 1 << 20).await?, + view.body(*id, 1 << 20).await? + ); + assert!(matches!( + workspace.body(*id, expected.size as usize - 1).await, + Err(ServingReadError::TooLarge) + )); + assert!(matches!( + workspace.body(*id, 65 << 20).await, + Err(ServingReadError::TooLarge) + )); + let wrong = + ObjectId::try_from(vec![1; if format == ObjectFormat::Sha1 { 32 } else { 20 }])?; + assert!(matches!( + workspace.contains(&[wrong]).await, + Err(ServingReadError::Context) + )); + assert!(matches!( + view.workspace(&[*id, *id], WorkspaceLimits::default()) + .await, + Err(ServingReadError::Context) + )); + assert!(matches!( + view.workspace(&[], WorkspaceLimits::default()).await, + Err(ServingReadError::Context) + )); + assert!(matches!( + view.workspace( + &[*id], + WorkspaceLimits { + cache_kib: 257, + ..WorkspaceLimits::default() + } + ) + .await, + Err(ServingReadError::Context) + )); + assert!(matches!( + view.workspace(&[missing(&f)?], WorkspaceLimits::default()) + .await, + Err(ServingReadError::Context) + )); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + assert!(matches!( + workspace.contains(&[*id]).await, + Err(ServingReadError::Inactive) + )); + assert!(matches!( + workspace.body(*id, 1 << 20).await, + Err(ServingReadError::Inactive) + )); + drop((view, workspace)); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn cancelled_construction_keeps_pin_and_admission_until_suspended_provider_drains() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = super::body::catalog(&f, provider.clone()).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("owner".into())).await?; + let oid = native + .fixture + .objects + .iter() + .find(|(_, (o, _))| o.kind == ObjectKind::Commit) + .ok_or("commit")? + .0; + let oid = *oid; + assert!(view.headers(&[oid]).await?[0].is_some()); + provider.armed.store(true, Ordering::Release); + let observer = + tokio::spawn(async move { view.workspace(&[oid], WorkspaceLimits::default()).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + observer.abort(); + assert!(observer.await.err().ok_or("cancelled")?.is_cancelled()); + let mut drain = tokio::spawn({ + let pool = pool.clone(); + async move { pool.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 1); + provider.proceed.add_permits(1); + timeout(Duration::from_secs(8), drain).await??; + super::pool::finish(&f, &pool, &q, tasks).await?; + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 0); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn closed_producer_renews_during_long_construction_and_returned_workspace_lifetime() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = super::body::catalog(&f, provider.clone()).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, _) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + // Warm immutable metadata before acquiring the short lease. Cold file + // hashing is covered by the separate construction/cancellation cases. + let oid = *native.fixture.objects.keys().next().ok_or("object")?; + let warm = + ServingOwner::start(ctx.clone(), q.clone(), f.begin([118; 16]), identity()?).await?; + timeout(Duration::from_secs(8), async { + while warm.stats().phase != ServingOwnerPhase::Ready { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let view = warm.snapshot(Some("owner".into())).await?; + assert!(view.headers(&[oid]).await?[0].is_some()); + drop(view); + assert_eq!( + warm.close_and_drain().await.phase, + ServingOwnerPhase::Released + ); + let mut input = f.begin([119; 16]); + input.lease_ms = 1000; + let owner = ServingOwner::start(ctx, q.clone(), input, identity()?).await?; + timeout(Duration::from_secs(8), async { + while owner.stats().phase != ServingOwnerPhase::Ready { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let view = owner.snapshot(Some("owner".into())).await?; + assert!( + view.headers(&[oid]) + .await + .map_err(|e| format!("warm short-lease headers: {e}"))?[0] + .is_some() + ); + provider.armed.store(true, Ordering::Release); + let observer = + tokio::spawn(async move { view.workspace(&[oid], WorkspaceLimits::default()).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + owner.close(); + tokio::time::sleep(Duration::from_millis(1500)).await; + timeout(Duration::from_secs(8), async { + while owner.stats().renewals < 2 { + assert_eq!( + owner.stats().phase, + ServingOwnerPhase::Ready, + "{:?}", + owner.stats() + ); + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert!(owner.stats().renewals >= 2, "{:?}", owner.stats()); + provider.proceed.add_permits(1); + let workspace = timeout(Duration::from_secs(8), observer) + .await?? + .map_err(|e| format!("renewed constructor: {e}"))?; + // Construction's final fresh authority check proves the complete result; + // short body/membership reads are exercised with their own deadline tests. + assert!(workspace.stats().objects > 0); + assert_eq!(pin_count(&f).await?, 1); + let mut drain = tokio::spawn({ + let owner = owner.clone(); + async move { owner.close_and_drain().await } + }); + assert!( + timeout(Duration::from_millis(50), &mut drain) + .await + .is_err() + ); + drop(workspace); + assert_eq!( + timeout(Duration::from_secs(8), drain).await??.phase, + ServingOwnerPhase::Released + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn reachability_stops_at_live_refs_without_scanning_other_history() -> Result { + use crate::git_gateway::{GatewayError, GitGateway}; + use crate::packs::ref_state::{RefStateSnapshot, RefStateSnapshotRoot}; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(InMemory::new()); + let fixture = prepare(format, provider.clone(), f.repository, 600) + .await + .map_err(|e| e.to_string())?; + let store = Arc::new(ArtifactStore::new(provider, f.repository)); + initialize(&f, store.clone()).await?; + f.install_generation(2, fixture.catalog, Some(fixture.refs)) + .await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, _) = super::body::serving_context(&f, store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let old = pool.snapshot(Some("owner".into())).await?; + let workspace = old.ref_workspace(WorkspaceLimits::default()).await?; + assert!(fixture.edges.len() as u64 > workspace.stats().objects + 500); + let wants = BTreeSet::from([fixture.main, fixture.root, fixture.side, fixture.tree]); + GitGateway::validate_wants(&workspace, &wants).await?; + for unrelated in [fixture.tag, fixture.wide.ok_or("wide")?, missing(&f)?] { + assert!(matches!( + GitGateway::validate_wants(&workspace, &BTreeSet::from([unrelated])).await, + Err(GatewayError::UnreachableWant) + )); + } + let refs = crate::packs::metadata::tests::git( + &workspace.git_dir(), + &["for-each-ref", "--format=%(refname)"], + None, + ) + .await?; + assert_eq!( + String::from_utf8(refs)?, + "refs/heads/main\nrefs/heads/side\n" + ); + let empty = RefStateSnapshotRoot::upload( + &store, + operation(122), + RefStateSnapshot { + repository: f.repository, + format, + generation: 2, + default_branch: "refs/heads/main".into(), + root: None, + }, + ) + .await?; + f.install_generation(3, fixture.catalog, Some(empty)) + .await?; + let current = pool.snapshot(Some("owner".into())).await?; + let empty_workspace = current.ref_workspace(WorkspaceLimits::default()).await?; + assert_eq!(empty_workspace.stats().objects, 0); + assert_eq!(empty_workspace.stats().packs, 0); + assert!(matches!( + GitGateway::validate_wants(&empty_workspace, &wants).await, + Err(GatewayError::UnreachableWant) + )); + GitGateway::validate_wants(&workspace, &wants).await?; // admitted old snapshot retains its roots + drop((old, current, workspace, empty_workspace)); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn returned_workspace_releases_read_credit_while_retaining_physical_generation() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let (native, _) = super::body::catalog(&f, Arc::new(InMemory::new())).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (_, files) = super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let ctx = ServingContext::new( + f.client(), + f.target.clone(), + f.authority(), + Arc::new(CatalogIndexes::new(native.store.clone(), f.format)), + files, + ServingReadBudget::new(2, tasks.clone())?, + "owner".into(), + )?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("owner".into())).await?; + let oid = *native.fixture.objects.keys().next().ok_or("object")?; + let workspace = view.workspace(&[oid], WorkspaceLimits::default()).await?; + assert_eq!(workspace.contains(&[oid]).await?, [true]); + assert!(workspace.body(oid, 1 << 20).await?.is_some()); + drop(view); + assert_eq!(workspace.contains(&[oid]).await?, [true]); + assert_eq!(pin_count(&f).await?, 1); + drop(workspace); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn suspended_construction_refuses_revoked_access_after_real_transfer_finishes() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = super::body::catalog(&f, provider.clone()).await?; + edit( + &f, + "INSERT INTO repository_members(account,role) VALUES('viewer','read')", + ) + .await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("viewer".into())).await?; + let oid = *native.fixture.objects.keys().next().ok_or("object")?; + assert!(view.headers(&[oid]).await?[0].is_some()); + provider.armed.store(true, Ordering::Release); + let observer = + tokio::spawn(async move { view.workspace(&[oid], WorkspaceLimits::default()).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + edit(&f, "DELETE FROM repository_members WHERE account='viewer'").await?; + assert_eq!(pin_count(&f).await?, 1); + provider.proceed.add_permits(1); + assert!(matches!( + timeout(Duration::from_secs(8), observer).await??, + Err(ServingReadError::Inactive) + )); + super::pool::finish(&f, &pool, &q, tasks).await?; + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 0); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn expired_lease_does_not_resurrect_or_release_suspended_construction() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let provider = Arc::new(super::blocked::Gate::new()); + let (native, _) = super::body::catalog(&f, provider.clone()).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, files) = + super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let mut input = f.begin([123; 16]); + input.lease_ms = 1000; + let owner = ServingOwner::start(ctx, q.clone(), input, identity()?).await?; + timeout(Duration::from_secs(8), async { + while owner.stats().phase != ServingOwnerPhase::Ready { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let view = owner.snapshot(Some("owner".into())).await?; + let oid = *native.fixture.objects.keys().next().ok_or("object")?; + assert!(view.headers(&[oid]).await?[0].is_some()); + let (dispatch, entered) = q.pause_for_test().await; + provider.armed.store(true, Ordering::Release); + let observer = + tokio::spawn(async move { view.workspace(&[oid], WorkspaceLimits::default()).await }); + timeout(Duration::from_secs(8), provider.entered.acquire()) + .await?? + .forget(); + timeout(Duration::from_secs(8), entered).await??; + tokio::time::sleep(Duration::from_millis(1100)).await; + assert_eq!(owner.stats().renewals, 0); + assert_eq!(pin_count(&f).await?, 1); + assert_eq!(files.native_stats()?.ok_or("stats")?.open_files, 1); + provider.proceed.add_permits(1); + assert!(matches!( + timeout(Duration::from_secs(8), observer).await??, + Err(ServingReadError::Inactive) + )); + dispatch.send(()).map_err(|_| "dispatcher lost")?; + assert_eq!( + timeout(Duration::from_secs(8), owner.close_and_drain()) + .await? + .phase, + ServingOwnerPhase::Released + ); + assert_eq!(pin_count(&f).await?, 0); + assert!(q.close_and_drain().await.is_empty()); + tasks.close(); + tasks.wait().await; + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn joint_ref_pages_stream_ten_thousand_names_without_an_object_root_limit() -> Result { + use crate::packs::ref_state::{ + RefStateRecord, RefStateSnapshot, RefStateSnapshotRoot, RefStateTree, + }; + let f = Fixture::new(ObjectFormat::Sha256).await?; + let (native, catalog) = super::body::catalog(&f, Arc::new(InMemory::new())).await?; + let oid = *native + .fixture + .objects + .iter() + .find(|(_, (o, _))| o.kind == ObjectKind::Blob) + .ok_or("blob")? + .0; + let refs = RefStateTree::new(native.store.clone(), f.format) + .build_sorted( + operation(124), + (0..10_000).map(|n| { + RefStateRecord::new( + &format!("refs/tags/item-{n:05}"), + crate::RefExpectation { + oid: Some(oid), + version: 1, + }, + f.format, + ) + }), + ) + .await?; + let refs = RefStateSnapshotRoot::upload( + &native.store, + operation(125), + RefStateSnapshot { + repository: f.repository, + format: f.format, + generation: 2, + default_branch: "refs/heads/main".into(), + root: refs, + }, + ) + .await?; + f.install_generation(3, catalog, Some(refs)).await?; + let q = super::pool::queue(&f)?; + let root = tempfile::TempDir::new()?; + let tasks = TaskTracker::new(); + let (ctx, _) = super::body::serving_context(&f, native.store.clone(), &root, tasks.clone())?; + let pool = ServingPool::new(ctx, q.clone(), ServingPoolLimits::default())?; + let view = pool.snapshot(Some("owner".into())).await?; + let workspace = timeout( + Duration::from_secs(30), + view.ref_workspace(WorkspaceLimits::default()), + ) + .await??; + assert_eq!(workspace.stats().objects, 1); + assert_eq!(workspace.stats().packs, 1); + let names = crate::packs::metadata::tests::git( + &workspace.git_dir(), + &["for-each-ref", "--format=%(refname)"], + None, + ) + .await?; + let names = String::from_utf8(names)?; + assert_eq!(names.lines().count(), 10_000); + assert_eq!(names.lines().next(), Some("refs/tags/item-00000")); + assert_eq!(names.lines().last(), Some("refs/tags/item-09999")); + drop((view, workspace)); + super::pool::finish(&f, &pool, &q, tasks).await?; + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/staged_durable.rs b/crates/canopy-server/src/packs/publication/tests/staged_durable.rs index d5099a3d..00bb6956 100644 --- a/crates/canopy-server/src/packs/publication/tests/staged_durable.rs +++ b/crates/canopy-server/src/packs/publication/tests/staged_durable.rs @@ -57,7 +57,11 @@ pub(super) async fn qualify( .ready_root_refusal(identity()?, store, root, budget.clone(), None) .await?, ); - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let mut head = None; let mut terminal = None; for offset in [0, 128, 256] { @@ -102,7 +106,9 @@ pub(super) async fn qualify( assert_eq!(failure.original.evidence_for_test(), evidence); assert_eq!(failure.registered.evidence(), &evidence); } - let cold = registered.clone().ready(f.client(), store.clone())?; + let cold = registered + .clone() + .ready(f.client(), store.clone(), f.authority())?; let failure = ticket .publish(&queue, cold) .err() @@ -141,7 +147,7 @@ pub(super) async fn qualify( let (release, wait) = tokio::sync::oneshot::channel(); let (entered, running) = tokio::sync::oneshot::channel(); let worker = if offset == 0 { - Some(ticket.spawn_bound(move |_| async move { + Some(ticket.spawn_bound(move |_, _context| async move { let _ = entered.send(()); wait.await.map_err(|_| StagingError::Worker)?; Ok(42u64) @@ -233,6 +239,7 @@ pub(super) async fn qualify( .dispatch_any( &f.client(), store, + &f.authority(), &std::sync::atomic::AtomicBool::new(false), ) .await?; @@ -263,7 +270,7 @@ pub(super) async fn qualify( let disk = budget.clone(); let mutation = identity()?; ticket - .spawn_bound(move |_| async move { + .spawn_bound(move |_, _context| async move { let guard = policy .ready(&owner) .await @@ -321,7 +328,7 @@ pub(super) async fn qualify( .await? .ok_or("durable terminal record")?; assert_eq!(loaded.evidence(), &evidence); - let actual = loaded.dispatch(&f.client(), store).await?; + let actual = loaded.dispatch(&f.client(), store, &f.authority()).await?; assert_eq!( (&actual.output, actual.receipt), (&value.output, value.receipt) @@ -386,7 +393,11 @@ pub(super) async fn qualify_fence(context: Context<'_>) -> Result { let registered = page.persist_recovery(store, identity()?, None).await?; let bound = page.bind_recovery(registered, store)?; session.fence(); - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let observer = queue .submit(bound) .await @@ -433,7 +444,11 @@ pub(super) async fn qualify_revoked(context: Context<'_>, root_case: bool) -> Re .ready_root_refusal(identity()?, store, root, budget.clone(), None) .await?, ); - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let mut head = None; let mut observer = None; for offset in [0, 128, 256] { @@ -474,7 +489,7 @@ pub(super) async fn qualify_revoked(context: Context<'_>, root_case: bool) -> Re let disk = budget; let mutation = identity()?; let ready = ticket - .spawn_bound(move |_| async move { + .spawn_bound(move |_, _context| async move { let guard = intent .ready(&owner) .await diff --git a/crates/canopy-server/src/packs/publication/tests/staging.rs b/crates/canopy-server/src/packs/publication/tests/staging.rs index 1bca2e25..102ed2f8 100644 --- a/crates/canopy-server/src/packs/publication/tests/staging.rs +++ b/crates/canopy-server/src/packs/publication/tests/staging.rs @@ -373,7 +373,8 @@ async fn physical_inputs_verified_before_binding_feed_the_existing_catalog_proof check(staged.token), Arc::clone(&indexes), Arc::clone(&files), - None + None, + fixture.authority(), ) .await, Err(PreparationBaseError::Inactive) @@ -397,6 +398,7 @@ async fn physical_inputs_verified_before_binding_feed_the_existing_catalog_proof indexes, files, Some(bound.receipt), + fixture.authority(), ) .await?, ); diff --git a/crates/canopy-server/src/packs/publication/tests/staging_receipt.rs b/crates/canopy-server/src/packs/publication/tests/staging_receipt.rs index 0bdc0bee..8649459f 100644 --- a/crates/canopy-server/src/packs/publication/tests/staging_receipt.rs +++ b/crates/canopy-server/src/packs/publication/tests/staging_receipt.rs @@ -82,12 +82,19 @@ async fn initial_staging_receipt_cold_restore_requires_actual_claim_and_keeps_or let input = f.begin([217; 16]); let mut mutation = identity()?; mutation.expires_at_ms = mutation.issued_at_ms + 1_000; - let command = f - .client() - .prepare_command::(&f.target, mutation, input.clone()) - .await?; + let command = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginStaging(input.clone()), + mutation, + ) + .await?; let evidence = command.evidence().clone(); - let original = command.execute().await?; + let original = command + .register(&f.client(), identity()?) + .await? + .recover_staging(&f.client()) + .await?; let StagingReply::Granted(old) = original.output else { return Err("missing admission".into()); }; @@ -113,7 +120,8 @@ async fn initial_staging_receipt_cold_restore_requires_actual_claim_and_keeps_or .await, PreparationDenial::Stale, ); - let coordinator = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let coordinator = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let ticket = coordinator .submit( saved @@ -146,10 +154,17 @@ async fn initial_staging_receipt_survives_reaping_but_does_not_restore_expired_c let f = Fixture::new(format).await?; let mut input = f.begin([218; 16]); input.lease_ms = 1_000; - let original = f - .client() - .command::(&f.target, identity()?, input.clone()) - .await?; + let original = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginStaging(input.clone()), + identity()?, + ) + .await? + .register(&f.client(), identity()?) + .await? + .recover_staging(&f.client()) + .await?; let StagingReply::Granted(lease) = original.output else { return Err("missing admission".into()); }; @@ -171,7 +186,8 @@ async fn initial_staging_receipt_survives_reaping_but_does_not_restore_expired_c .ok_or("reaped receipt missing")?; assert_eq!(saved.receipt(), original.receipt); assert_eq!(saved.lease(), *lease); - let coordinator = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let coordinator = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let ticket = coordinator .submit( saved @@ -213,7 +229,8 @@ async fn initial_staging_receipt_is_internal_knowledge_after_write_revocation() for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { let f = Fixture::new(format).await?; let input = f.begin([215; 16]); - let coordinator = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let coordinator = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; coordinator.fault_for_test(2); let ticket = coordinator .submit( @@ -253,11 +270,12 @@ async fn initial_staging_receipt_is_internal_knowledge_after_write_revocation() } #[tokio::test] -async fn initial_staging_receipt_corruption_keeps_original_evidence_and_reservation() -> Result { +async fn staging_custody_intent_corruption_keeps_original_evidence_and_reservation() -> Result { for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { let f = Fixture::new(format).await?; let input = f.begin([214; 16]); - let coordinator = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let coordinator = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; coordinator.fault_for_test(2); let ticket = coordinator .submit( @@ -276,19 +294,15 @@ async fn initial_staging_receipt_corruption_keeps_original_evidence_and_reservat return Err("missing evidence".into()); }; let evidence = (**evidence).clone(); - let saved = StagingAdmission::load(&f.client(), &f.target, input.operation) + let saved = RegisteredCustody::load_latest(&f.client(), &f.target, input.operation) .await? - .ok_or("receipt missing")?; - let original = saved.original(&evidence)?.ok_or("original stamp differs")?; - let other = f - .client() - .prepare_command::(&f.target, identity()?, input) - .await?; - assert!(saved.original(other.evidence())?.is_none()); + .ok_or("intent missing")?; + assert_eq!(saved.evidence(), &evidence); + let original = saved.recover_staging(&f.client()).await?; let body = f .handle - .query(0, 2048, |db| { - Ok(db.query_row("SELECT initial_staging FROM pushes", [], |r| { + .query(0, 4096, |db| { + Ok(db.query_row("SELECT intent FROM catalog_custody_commands WHERE operation = x'd6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6'", [], |r| { r.get::<_, Vec>(0) })?) }) @@ -296,11 +310,11 @@ async fn initial_staging_receipt_corruption_keeps_original_evidence_and_reservat let mut corrupt = body.clone(); let end = corrupt.last_mut().ok_or("empty receipt")?; *end ^= 1; - edit(&f, "DROP TRIGGER push_initial_staging_immutable").await?; + edit(&f, "DROP TRIGGER catalog_custody_identity_immutable").await?; edit( &f, &format!( - "UPDATE pushes SET initial_staging=x'{}'", + "UPDATE catalog_custody_commands SET intent=x'{}' WHERE step=0", hex::encode(corrupt) ), ) @@ -311,14 +325,17 @@ async fn initial_staging_receipt_corruption_keeps_original_evidence_and_reservat else { return Err("corruption lost uncertainty".into()); }; - let StagingError::ReceiptRecovery { + let StagingError::Custody { evidence: retained, .. } = error.as_ref() else { return Err("missing receipt recovery error".into()); }; assert_eq!(retained.as_ref(), &evidence); - assert_eq!(coordinator.stats().command_bytes, 8192); + assert_eq!( + coordinator.stats().command_bytes, + super::super::custody::RESERVATION + ); assert!(matches!( ticket.spawn(|_| async { Ok(()) }), Err(StagingError::Inactive) @@ -326,7 +343,10 @@ async fn initial_staging_receipt_corruption_keeps_original_evidence_and_reservat assert_eq!(f.counts().await?, (1, 1)); edit( &f, - &format!("UPDATE pushes SET initial_staging=x'{}'", hex::encode(body)), + &format!( + "UPDATE catalog_custody_commands SET intent=x'{}' WHERE step=0", + hex::encode(body) + ), ) .await?; coordinator.recover(&ticket)?; diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service.rs b/crates/canopy-server/src/packs/publication/tests/staging_service.rs index a1a76e1e..94f00cf3 100644 --- a/crates/canopy-server/src/packs/publication/tests/staging_service.rs +++ b/crates/canopy-server/src/packs/publication/tests/staging_service.rs @@ -1,5 +1,8 @@ mod bound; +mod physical; mod publication; +pub(super) mod restore; +mod retirement; use super::*; use tokio::{ sync::oneshot, @@ -44,6 +47,134 @@ async fn changed_lease(ticket: &StagingTicket, old: i64) -> Result .await? } +#[tokio::test] +async fn staged_service_registrar_loss_retains_both_commands_through_cancellation_and_close() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [4, 5, 6] { + let f = Fixture::new(format).await?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + c.fault_for_test(fault); + let ticket = submit(&f, &c, [fault + 120; 16], "owner").await?; + let StagingState::Uncertain(error) = terminal(&ticket).await? else { + return Err("registrar loss not retained".into()); + }; + let (original, registrar) = ticket + .custody_evidence_for_test() + .ok_or("custody command missing")?; + let registrar = registrar.ok_or("registrar missing")?; + assert_ne!(original, registrar); + if fault != 6 { + let StagingError::Custody { evidence, source } = &*error else { + return Err("registrar source lost".into()); + }; + assert_eq!(**evidence, original); + let CustodyError::Registration(source) = &**source else { + return Err("wrong protocol phase".into()); + }; + let InvocationError::Pending(evidence) = &**source else { + return Err("registrar original lost".into()); + }; + assert_eq!(**evidence, registrar); + } + assert!(matches!( + f.client().resolve(&original).await?, + cellule_runtime::Resolution::Absent + )); + assert_eq!(f.counts().await?, (0, 0)); + assert_eq!(c.stats().command_bytes, super::super::custody::RESERVATION); + drop(ticket); + let retained = c + .pending([fault + 120; 16]) + .ok_or("observer dropped admitted work")?; + assert_eq!(c.close_and_drain().await.len(), 1); + assert_eq!( + retained.custody_evidence_for_test(), + Some((original.clone(), Some(registrar.clone()))) + ); + c.recover(&retained)?; + assert!(matches!(terminal(&retained).await?, StagingState::Stopped)); + let saved = RegisteredCustody::load_latest(&f.client(), &f.target, [fault + 120; 16]) + .await? + .ok_or("registered original missing")?; + assert_eq!(saved.evidence(), &original); + let result = saved.recover_staging(&f.client()).await?; + assert!(matches!(result.output, StagingReply::Granted(_))); + assert!(matches!( + f.client().resolve(®istrar).await?, + cellule_runtime::Resolution::Committed(_) + )); + assert_eq!(f.counts().await?, (1, 1)); + assert_eq!(c.stats().admitted, 0); + assert_eq!(c.stats().command_bytes, 0); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn staged_service_renew_and_bind_registrar_loss_never_replaces_either_identity() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [4, 5, 6] { + let f = Fixture::new(format).await?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let ticket = submit(&f, &c, [fault + 130; 16], "owner").await?; + let staged = active(&ticket).await?; + c.fault_for_test(fault); + ticket.renew_for_test(); + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + let renewal = ticket + .custody_evidence_for_test() + .ok_or("renewal originals lost")?; + assert!(matches!( + f.client().resolve(&renewal.0).await?, + cellule_runtime::Resolution::Absent + )); + c.recover(&ticket)?; + assert_eq!(active(&ticket).await?.token, staged.token); + let registered = + RegisteredCustody::load_latest(&f.client(), &f.target, staged.token.operation) + .await? + .ok_or("renewal registration missing")?; + assert_eq!(registered.evidence(), &renewal.0); + c.fault_for_test(fault); + ticket.seal()?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + let binding = ticket + .custody_evidence_for_test() + .ok_or("bind originals lost")?; + assert_ne!(binding, renewal); + c.recover(&ticket)?; + let StagingState::Bound(bound) = terminal(&ticket).await? else { + return Err("bind not recovered".into()); + }; + let registered = + RegisteredCustody::load_latest(&f.client(), &f.target, staged.token.operation) + .await? + .ok_or("bind registration missing")?; + assert_eq!(registered.evidence(), &binding.0); + assert_eq!( + registered.recover_preparation(&f.client()).await?.receipt, + bound.receipt + ); + assert_eq!(bound.lease.token, staged.token); + assert_eq!(f.counts().await?, (1, 1)); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + #[tokio::test] async fn staged_service_recovers_original_begin_after_sdk_expiry_before_allowing_uploads() -> Result { @@ -54,8 +185,11 @@ async fn staged_service_recovers_original_begin_after_sdk_expiry_before_allowing (ObjectFormat::Sha256, 3), ] { let fixture = Fixture::new(format).await?; - let coordinator = - StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let request = fixture.begin([219; 16]); let mut mutation = identity()?; mutation.expires_at_ms = mutation.issued_at_ms + 2_000; @@ -73,7 +207,10 @@ async fn staged_service_recovers_original_begin_after_sdk_expiry_before_allowing return Err("missing original Begin evidence".into()); }; let evidence = (**evidence).clone(); - assert_eq!(coordinator.stats().command_bytes, 8192); + assert_eq!( + coordinator.stats().command_bytes, + super::super::custody::RESERVATION + ); assert!(matches!( ticket.spawn(|_| async { Ok(()) }), Err(StagingError::Inactive) @@ -84,7 +221,9 @@ async fn staged_service_recovers_original_begin_after_sdk_expiry_before_allowing return Err("lost acknowledgement did not follow an accepted Begin".into()); }; let mut decoder = BoundedDecoder::new(original.result(), 4096)?; - let StagingReply::Granted(lease) = StagingReply::decode(&mut decoder)? else { + let CustodyReply::Staging(StagingReply::Granted(lease)) = + CustodyReply::decode(&mut decoder)? + else { return Err("original admission was not granted".into()); }; decoder.finish()?; @@ -141,7 +280,11 @@ async fn staged_service_recovers_original_begin_after_sdk_expiry_before_allowing async fn staged_service_canceled_observers_keep_workers_and_results_until_single_handoff() -> Result { let fixture = Fixture::new(ObjectFormat::Sha256).await?; - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ticket = submit(&fixture, &coordinator, [220; 16], "owner").await?; let initial = active(&ticket).await?; let (release, wait) = oneshot::channel(); @@ -193,8 +336,11 @@ async fn staged_service_resolves_begin_renew_and_bind_exactly_after_absence_lost -> Result { for fault in [1, 2, 3] { let fixture = Fixture::new(ObjectFormat::Sha256).await?; - let coordinator = - StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; coordinator.fault_for_test(fault); let ticket = submit(&fixture, &coordinator, [221; 16], "owner").await?; assert!(matches!( @@ -202,7 +348,10 @@ async fn staged_service_resolves_begin_renew_and_bind_exactly_after_absence_lost StagingState::Uncertain(_) )); assert_eq!(coordinator.stats().admitted, 1); - assert_eq!(coordinator.stats().command_bytes, 8192); + assert_eq!( + coordinator.stats().command_bytes, + super::super::custody::RESERVATION + ); coordinator.recover(&ticket)?; let original = active(&ticket).await?; assert_eq!(original.token.artifact_operation, artifact_number(1)); @@ -231,7 +380,10 @@ async fn staged_service_resolves_begin_renew_and_bind_exactly_after_absence_lost let evidence = (**evidence).clone(); let pending = timeout(Duration::from_secs(10), coordinator.close_and_drain()).await?; assert_eq!(pending.len(), 1); - assert_eq!(coordinator.stats().command_bytes, 8192); + assert_eq!( + coordinator.stats().command_bytes, + super::super::custody::RESERVATION + ); coordinator.recover(&pending[0])?; let StagingState::Bound(bound) = terminal(&pending[0]).await? else { return Err("bind did not resolve".into()); @@ -259,7 +411,11 @@ async fn staged_service_replayed_renewal_is_not_a_new_clock_or_permission_after_ "INSERT INTO repository_members VALUES('writer','write')", ) .await?; - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ticket = submit(&fixture, &coordinator, [222; 16], "writer").await?; active(&ticket).await?; let (entered, started) = oneshot::channel(); @@ -311,6 +467,7 @@ async fn staged_service_account_operation_and_worker_bounds_preserve_rejected_re workers_per_actor: 1, ..StagingLimits::default() }, + fixture.authority(), )?; let first = submit(&fixture, &coordinator, [223; 16], "owner").await?; active(&first).await?; @@ -368,8 +525,11 @@ async fn staged_service_worker_failure_and_panic_fence_before_binding_and_releas -> Result { for panic in [false, true] { let fixture = Fixture::new(ObjectFormat::Sha1).await?; - let coordinator = - StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ticket = submit(&fixture, &coordinator, [226; 16], "owner").await?; let lease = active(&ticket).await?; let work = ticket.spawn(move |_| async move { @@ -405,8 +565,11 @@ async fn staged_service_owned_native_verification_hands_off_to_the_existing_priv use cellule_ltx::DiskBudget; for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { let fixture = Fixture::new(format).await?; - let coordinator = - StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ticket = submit(&fixture, &coordinator, [227; 16], "owner").await?; let initial = active(&ticket).await?; let provider: Arc = Arc::new(InMemory::new()); @@ -454,7 +617,7 @@ async fn staged_service_owned_native_verification_hands_off_to_the_existing_priv let bound_work = { let root = root.clone(); let budget = budget.clone(); - ticket.spawn_bound(move |_| async move { + ticket.spawn_bound(move |_, _context| async move { async { let mut assembler = CatalogPreparation::new(root.path(), budget.clone(), base, limits()) @@ -495,6 +658,7 @@ async fn staged_service_automatic_renewal_runs_without_an_observer_or_manual_tic renew_before_ms: DEFAULT_LEASE_MS - 1000, ..StagingLimits::default() }, + fixture.authority(), )?; let ticket = submit(&fixture, &coordinator, [228; 16], "owner").await?; let lease = active(&ticket).await?; @@ -569,11 +733,15 @@ async fn staged_service_rejects_invalid_profiles_foreign_targets_and_duplicate_l }, ] { assert!(matches!( - StagingCoordinator::new(fixture.target.clone(), limits), + StagingCoordinator::new(fixture.target.clone(), limits, fixture.authority(),), Err(StagingError::InvalidLimits) )); } - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ready = ReadyStaging::new( fixture.client(), fixture.target.clone(), @@ -599,7 +767,11 @@ async fn staged_service_rejects_invalid_profiles_foreign_targets_and_duplicate_l .ok_or("duplicate accepted")?; assert!(matches!(error, StagingError::Duplicate)); let foreign = Fixture::new(ObjectFormat::Sha256).await?; - let other = StagingCoordinator::new(foreign.target.clone(), StagingLimits::default())?; + let other = StagingCoordinator::new( + foreign.target.clone(), + StagingLimits::default(), + foreign.authority(), + )?; let (error, _) = other.submit(ready).err().ok_or("foreign accepted")?; assert!(matches!(error, StagingError::Foreign)); assert!(matches!(other.recover(&ticket), Err(StagingError::Foreign))); @@ -643,7 +815,11 @@ async fn staged_service_revocation_drops_completed_owned_results_before_releasin "INSERT INTO repository_members VALUES('writer','write')", ) .await?; - let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits::default(), + fixture.authority(), + )?; let ticket = submit(&fixture, &coordinator, [231; 16], "writer").await?; active(&ticket).await?; let dropped = Arc::new(AtomicBool::new(false)); diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service/bound.rs b/crates/canopy-server/src/packs/publication/tests/staging_service/bound.rs index 97951482..2e3b38f9 100644 --- a/crates/canopy-server/src/packs/publication/tests/staging_service/bound.rs +++ b/crates/canopy-server/src/packs/publication/tests/staging_service/bound.rs @@ -50,13 +50,7 @@ pub(super) async fn claim( Ok(c.submit(ready).map_err(|(e, _)| e)?) } async fn new_token(f: &Fixture, op: [u8; 16]) -> Result { - Ok(lease( - f.client() - .command::(&f.target, identity()?, f.begin(op)) - .await? - .output, - )? - .token) + Ok(lease(registered_preparation(f, op).await?.output)?.token) } #[tokio::test] async fn bound_service_automatic_renewal_keeps_canceled_worker_and_result_owned_through_close() @@ -68,6 +62,7 @@ async fn bound_service_automatic_renewal_keeps_canceled_worker_and_result_owned_ renew_before_ms: DEFAULT_LEASE_MS - 1000, ..StagingLimits::default() }, + f.authority(), )?; let ticket = bind(&f, &c, [203; 16], "owner").await?; let original = ticket.bound_result().ok_or("binding receipt")?; @@ -75,7 +70,7 @@ async fn bound_service_automatic_renewal_keeps_canceled_worker_and_result_owned_ let weak = Arc::downgrade(&session); let (release, wait) = oneshot::channel(); let (entered, running) = oneshot::channel(); - let worker = ticket.spawn_bound(move |session| async move { + let worker = ticket.spawn_bound(move |session, _context| async move { session.live_lease()?; let _ = entered.send(()); wait.await.map_err(|_| StagingError::Worker)?; @@ -114,7 +109,11 @@ async fn bound_service_automatic_renewal_keeps_canceled_worker_and_result_owned_ }) .await?; assert!(!close.is_finished()); - assert!(retained.spawn_bound(|_| async { Ok(()) }).is_err()); + assert!( + retained + .spawn_bound(|_, _context| async { Ok(()) }) + .is_err() + ); release.send(()).map_err(|_| "worker lost")?; let result = retained .pending_task::>(id) @@ -141,7 +140,8 @@ async fn bound_service_renewal_retains_exact_absent_lost_and_panicked_commands_t for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { for fault in [1, 2, 3] { let f = Fixture::new(format).await?; - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let ticket = bind(&f, &c, [204; 16], "owner").await?; let original = ticket.bound_result().ok_or("binding")?; let shared = ticket.bound_session()?; @@ -168,7 +168,10 @@ async fn bound_service_renewal_retains_exact_absent_lost_and_panicked_commands_t other => return Err(format!("unexpected {other:?}").into()), }; assert_eq!(sequence.is_some(), fault != 1); - assert_eq!(c.stats().command_bytes, 8192); + assert_eq!( + c.stats().command_bytes, + super::super::super::custody::RESERVATION + ); drop(ticket); let retained = c.pending([204; 16]).ok_or("renewal lost")?; assert_eq!(c.close_and_drain().await.len(), 1); @@ -209,7 +212,8 @@ async fn bound_service_claim_retains_exact_identity_and_new_namespace_through_cl for fault in [1, 2, 3] { let f = Fixture::new(format).await?; let old = new_token(&f, [205; 16]).await?; - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let mutation = identity()?; c.fault_for_test(fault); let ticket = claim(&f, &c, old, mutation).await?; @@ -232,17 +236,11 @@ async fn bound_service_claim_retains_exact_identity_and_new_namespace_through_cl }; assert_ne!(bound.lease.token, old); assert_ne!(bound.lease.token.artifact_operation, old.artifact_operation); - let replay = f - .client() - .command::( - &f.target, - mutation, - LeaseRequest { - check: check(old), - lease_ms: DEFAULT_LEASE_MS, - }, - ) - .await?; + let saved = RegisteredCustody::load_latest(&f.client(), &f.target, old.operation) + .await? + .ok_or("claim intent missing")?; + assert_eq!(saved.evidence().identity(), mutation); + let replay = saved.recover_preparation(&f.client()).await?; assert_eq!(bound.receipt, replay.receipt); assert_eq!(bound.lease, lease(replay.output)?); let cellule_runtime::Resolution::Committed(value) = @@ -265,7 +263,8 @@ async fn bound_service_known_renewal_receipts_survive_revocation_expiry_and_supe for committed in [false, true] { for mode in [0, 1, 2] { let f = Fixture::new(ObjectFormat::Sha256).await?; - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let ticket = bind(&f, &c, [206; 16], "owner").await?; let session = ticket.bound_session()?; let binding = ticket.bound_result().ok_or("binding")?; @@ -352,14 +351,24 @@ async fn bound_service_phase_handoff_and_residence_cap_fence_existing_bases_and_ let native = crate::packs::catalog::tests::prepared_for_repository(f.format, f.repository).await?; f.install_catalog(1, native.stored).await?; - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let ticket = submit(&f, &c, [207; 16], "owner").await?; active(&ticket).await?; let work = ticket.spawn(|ctx| async { Ok(ctx) })?; let old = work.wait().await.map_err(|e| e.to_string())?; + let old_token = old.token()?; + // A context now owns its physical worker admission. Retain only the + // historical token before handoff; a live context must keep Bind blocked. + drop(old); ticket.seal()?; assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); - assert!(old.ensure_live().is_err()); + assert!( + f.client() + .query::(&f.target, None, check(old_token)) + .await? + .output + .is_none() + ); assert!(ticket.spawn(|_| async { Ok(()) }).is_err()); let session = ticket.bound_session()?; let store = native.store.clone(); @@ -395,7 +404,7 @@ async fn bound_service_phase_handoff_and_residence_cap_fence_existing_bases_and_ )); assert!(!session.fenced.load(std::sync::atomic::Ordering::Acquire)); let (entered, running) = oneshot::channel(); - let worker = ticket.spawn_bound(move |_| async move { + let worker = ticket.spawn_bound(move |_, _context| async move { let _ = entered.send(()); std::future::pending::>().await })?; @@ -440,6 +449,7 @@ async fn bound_service_worker_caps_results_and_failure_reuse_staging_admission() workers_per_actor: 1, ..StagingLimits::default() }, + f.authority(), )?; let a = bind(&f, &c, [208; 16], "owner").await?; let b = bind(&f, &c, [209; 16], "owner").await?; @@ -452,13 +462,13 @@ async fn bound_service_worker_caps_results_and_failure_reuse_staging_admission() wrong: wrong.clone(), }; let (done, completed) = oneshot::channel(); - let work = a.spawn_bound(move |_| async move { + let work = a.spawn_bound(move |_, _context| async move { let _ = done.send(()); Ok(owned) })?; timeout(Duration::from_secs(10), completed).await??; assert_eq!(c.stats().workers, 1); - assert!(b.spawn_bound(|_| async { Ok(()) }).is_err()); + assert!(b.spawn_bound(|_, _context| async { Ok(()) }).is_err()); super::super::publishing::edit( &f, "UPDATE repository_identity SET owner='other' WHERE singleton=1", @@ -479,7 +489,8 @@ async fn bound_service_worker_caps_results_and_failure_reuse_staging_admission() StagingLimits { bound_lifetime_ms: invalid, ..StagingLimits::default() - } + }, + f.authority(), ) .is_err() ); @@ -508,13 +519,14 @@ async fn bound_service_checkpoint_shares_renewal_order_exact_recovery_and_origin return Err("source bind".into()); }; assert!(source.close_and_drain().await.is_empty()); - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let ticket = claim(&f, &c, old.lease.token, identity()?).await?; assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); let session = ticket.bound_session()?; let parent = prior.clone(); let provider = store.clone(); - let worker = ticket.spawn_bound(move |s| async move { + let worker = ticket.spawn_bound(move |s, _context| async move { s.adopt_native_inputs(provider, &parent) .await .map_err(|e| StagingError::Input(Box::new(e))) @@ -551,7 +563,10 @@ async fn bound_service_checkpoint_shares_renewal_order_exact_recovery_and_origin return Err("checkpoint evidence".into()); }; let evidence = (**evidence).clone(); - assert_eq!(c.stats().command_bytes, 12 << 10); + assert_eq!( + c.stats().command_bytes, + super::super::super::custody::RESERVATION + 4096 + ); if fault == 2 { super::super::publishing::edit( &f, @@ -631,7 +646,7 @@ async fn bound_service_restored_owner_claim_retains_old_pin_and_owns_new_session ) .await?; let client = CellClient::local(f.registry.clone(), handle.clone()); - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; let mutation = identity()?; c.fault_for_test(2); let ready = ReadyStaging::claim_bound( @@ -655,21 +670,16 @@ async fn bound_service_restored_owner_claim_retains_old_pin_and_owns_new_session }; assert_ne!(bound.lease.token.owner, old.owner); assert_ne!(bound.lease.token.artifact_operation, old.artifact_operation); - let replay = client - .command::( - &f.target, - mutation, - LeaseRequest { - check: check(old), - lease_ms: DEFAULT_LEASE_MS, - }, - ) - .await?; + let saved = RegisteredCustody::load_latest(&client, &f.target, old.operation) + .await? + .ok_or("claim intent missing")?; + assert_eq!(saved.evidence().identity(), mutation); + let replay = saved.recover_preparation(&client).await?; assert_eq!(bound.receipt, replay.receipt); let old_pin = handle.query(0, 32, move |conn| { Ok(conn.query_row("SELECT generation FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", rusqlite::params![old.owner.incarnation.as_bytes().as_slice(), old.attempt as i64], |row| row.get::<_, i64>(0))?.to_be_bytes().to_vec()) }).await?; assert_eq!(old_pin.as_slice(), 0i64.to_be_bytes()); - let worker = - ticket.spawn_bound(|session| async move { Ok(session.live_lease()?.0.token) })?; + let worker = ticket + .spawn_bound(|session, _context| async move { Ok(session.live_lease()?.0.token) })?; assert_eq!( worker.wait().await.map_err(|e| e.to_string())?, bound.lease.token diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service/physical.rs b/crates/canopy-server/src/packs/publication/tests/staging_service/physical.rs new file mode 100644 index 00000000..971992a0 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/staging_service/physical.rs @@ -0,0 +1,116 @@ +//! Physical jobs retain original admission after the async producer is gone. +use super::*; + +// Always release a blocked test job, including after an assertion or timeout. +struct Release(Option>); +impl Drop for Release { + fn drop(&mut self) { + if let Some(sender) = self.0.take() { + let _ = sender.send(()); + } + } +} +fn held_worker( + context: StagingContext, +) -> (Release, oneshot::Receiver<()>, tokio::task::JoinHandle<()>) { + let (entered, running) = oneshot::channel(); + let (release, wait) = std::sync::mpsc::channel(); + let owner = context.physical_owner(); + let worker = tokio::task::spawn_blocking(move || { + let _owner = owner; + let _ = entered.send(()); + let _ = wait.recv(); + }); + (Release(Some(release)), running, worker) +} + +#[tokio::test] +async fn physical_creating_worker_prevents_bind_and_credit_reuse_after_result_transfer() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits { + workers: 1, + workers_per_actor: 1, + ..StagingLimits::default() + }, + f.authority(), + )?; + let ticket = submit(&f, &c, [233; 16], "owner").await?; + let lease = active(&ticket).await?; + let work = ticket.spawn(|context| async move { Ok(held_worker(context)) })?; + let (release, running, worker) = work.wait().await.map_err(|e| e.to_string())?; + timeout(Duration::from_secs(10), running).await??; + // The result was transferred and the producer has exited. Only the + // detached physical job owns the original worker admission now. + ticket.seal()?; + assert_eq!(c.stats().workers, 1); + assert!(matches!( + ticket.spawn(|_| async { Ok(()) }), + Err(StagingError::Capacity) + )); + assert!( + f.client() + .query::(&f.target, None, check(lease.token)) + .await? + .output + .is_none() + ); + assert!(!worker.is_finished()); + drop(release); + timeout(Duration::from_secs(10), worker).await??; + assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); + assert_eq!(c.stats().workers, 0); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn physical_bound_worker_outlives_custody_fence_async_abort_and_observer_drop() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let ticket = bound::bind(&f, &c, [234; 16], "owner").await?; + let session = ticket.bound_session()?; + let (physical, running) = oneshot::channel(); + let work = ticket.spawn_bound(move |_, context| async move { + let held = held_worker(context); + physical.send(held).map_err(|_| StagingError::Worker)?; + std::future::pending::>().await + })?; + let (release, entered, worker) = timeout(Duration::from_secs(10), running).await??; + timeout(Duration::from_secs(10), entered).await??; + drop(work); + ticket.expire_bound_for_test()?; + timeout(Duration::from_secs(10), async { + while !matches!(ticket.state(), StagingState::Fenced(_)) { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert!(session.live_lease().is_err()); + assert!(ticket.bound_session().is_err()); + assert_eq!(c.stats().workers, 1); + assert_eq!(c.stats().admitted, 1); + drop(ticket); + let closing = c.clone(); + let close = tokio::spawn(async move { closing.close_and_drain().await }); + timeout(Duration::from_secs(10), async { + while !c.stats().closed { + tokio::task::yield_now().await; + } + }) + .await?; + assert!(!close.is_finished()); + drop(release); + timeout(Duration::from_secs(10), worker).await??; + assert!(timeout(Duration::from_secs(10), close).await??.is_empty()); + assert_eq!(c.stats().workers, 0); + assert_eq!(c.stats().admitted, 0); + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service/publication.rs b/crates/canopy-server/src/packs/publication/tests/staging_service/publication.rs index fda38912..68e1eef6 100644 --- a/crates/canopy-server/src/packs/publication/tests/staging_service/publication.rs +++ b/crates/canopy-server/src/packs/publication/tests/staging_service/publication.rs @@ -30,9 +30,16 @@ async fn bound_final_waits_for_exact_renewal_and_adopted_checkpoint_recovery_bef return Err("source binding lost".into()); }; assert!(source.close_and_drain().await.is_empty()); - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; - let p = - PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits::default(), + f.authority(), + )?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let ticket = super::bound::claim(&f, &c, original.lease.token, identity()?).await?; assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); let session = ticket.bound_session()?; @@ -65,7 +72,10 @@ async fn bound_final_waits_for_exact_renewal_and_adopted_checkpoint_recovery_bef assert!(matches!(error.as_ref(), StagingError::Checkpoint(_))); } assert!(matches!(observer.state(), PublicationState::Held)); - assert_eq!(c.stats().command_bytes, 12 << 10); + assert_eq!( + c.stats().command_bytes, + super::super::super::custody::RESERVATION + 4096 + ); assert_eq!(p.stats().await.held, 1); assert_eq!(c.close_and_drain().await.len(), 1); assert_eq!(p.close_and_drain().await.len(), 1); @@ -121,14 +131,18 @@ async fn bound_final_publication_drains_retained_work_and_due_renewal_through_cl -> Result { for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { let f = Fixture::new(format).await?; - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let ticket = super::bound::bind(&f, &c, [231; 16], "owner").await?; let session = ticket.bound_session()?; let input = ready(&session).await?; let (release, blocked) = oneshot::channel(); let (entered, running) = oneshot::channel(); - let work = ticket.spawn_bound(move |_| async move { + let work = ticket.spawn_bound(move |_, _context| async move { let _ = entered.send(()); blocked.await.map_err(|_| StagingError::Worker)?; Ok(42u64) @@ -140,7 +154,7 @@ async fn bound_final_publication_drains_retained_work_and_due_renewal_through_cl drop(publication); drop(work); assert!(matches!(ticket.state(), StagingState::Finishing)); - assert!(ticket.spawn_bound(|_| async { Ok(()) }).is_err()); + assert!(ticket.spawn_bound(|_, _context| async { Ok(()) }).is_err()); assert!(ticket.bound_session().is_err()); assert_eq!(p.stats().await.held, 1); assert_eq!(c.stats().workers, 1); @@ -215,15 +229,15 @@ fn bound_final_exact_recovery_preserves_commits_and_refuses_absence_after_custod } async fn exact_case(format: ObjectFormat, fault: u8, expired: bool) -> Result { let f = Fixture::new(format).await?; - let c = StagingCoordinator::new( + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let p = PublicationCoordinator::new( f.target.clone(), - StagingLimits { - bound_lifetime_ms: 1000, - ..StagingLimits::default() - }, + PublicationLimits::default(), + f.publication_budget.clone(), )?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; let ticket = super::bound::bind(&f, &c, [232; 16], "owner").await?; + // Start the short ceiling in the publication phase, after Bind setup. + ticket.limit_bound_ceiling_for_test(Duration::from_millis(1000))?; let session = ticket.bound_session()?; p.fault_for_test(fault); let observer = ticket.publish(&p, ready(&session).await?)?; @@ -250,7 +264,10 @@ async fn exact_case(format: ObjectFormat, fault: u8, expired: bool) -> Result { other => return Err(format!("unexpected resolution {other:?}").into()), }; assert_eq!(original.is_some(), fault != 1); - assert_eq!(c.stats().command_bytes, 8 << 10); + assert_eq!( + c.stats().command_bytes, + super::super::super::custody::RESERVATION + ); assert_eq!(p.stats().await.command_bytes, 8 << 20); if expired { tokio::time::pause(); @@ -325,15 +342,15 @@ async fn bound_final_ceiling_discards_held_proof_and_drops_result_before_worker_ } } let f = Fixture::new(ObjectFormat::Sha256).await?; - let c = StagingCoordinator::new( + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let p = PublicationCoordinator::new( f.target.clone(), - StagingLimits { - bound_lifetime_ms: 1000, - ..StagingLimits::default() - }, + PublicationLimits::default(), + f.publication_budget.clone(), )?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; let ticket = super::bound::bind(&f, &c, [233; 16], "owner").await?; + // Start the short ceiling in the publication phase, after Bind setup. + ticket.limit_bound_ceiling_for_test(Duration::from_millis(1000))?; let session = ticket.bound_session()?; let input = ready(&session).await?; let dropped = Arc::new(AtomicBool::new(false)); @@ -344,7 +361,7 @@ async fn bound_final_ceiling_discards_held_proof_and_drops_result_before_worker_ wrong: wrong.clone(), }; let (done, completed) = oneshot::channel(); - let work = ticket.spawn_bound(move |_| async move { + let work = ticket.spawn_bound(move |_, _context| async move { let _ = done.send(()); Ok(owned) })?; @@ -375,8 +392,12 @@ async fn bound_final_ceiling_discards_held_proof_and_drops_result_before_worker_ async fn bound_final_refusals_keep_exact_ready_and_require_shared_session_and_final_kind() -> Result { let f = Fixture::new(ObjectFormat::Sha256).await?; - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let ticket = super::bound::bind(&f, &c, [234; 16], "owner").await?; let session = ticket.bound_session()?; let unrelated = Arc::new( @@ -385,6 +406,7 @@ async fn bound_final_refusals_keep_exact_ready_and_require_shared_session_and_fi f.target.clone(), check(session.lease.token), None, + f.authority(), ) .await?, ); @@ -409,6 +431,7 @@ async fn bound_final_refusals_keep_exact_ready_and_require_shared_session_and_fi f.repository, )?, PublicationLimits::default(), + f.publication_budget.clone(), )?; let failure = ticket .publish(&foreign, ready(&session).await?) @@ -445,15 +468,15 @@ async fn bound_final_refusals_keep_exact_ready_and_require_shared_session_and_fi #[tokio::test] async fn bound_final_queued_transport_rechecks_ceiling_before_initial_execution() -> Result { let f = Fixture::new(ObjectFormat::Sha256).await?; - let c = StagingCoordinator::new( + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let p = PublicationCoordinator::new( f.target.clone(), - StagingLimits { - bound_lifetime_ms: 1000, - ..StagingLimits::default() - }, + PublicationLimits::default(), + f.publication_budget.clone(), )?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; let ticket = super::bound::bind(&f, &c, [236; 16], "owner").await?; + // Start the short ceiling in the publication phase, after Bind setup. + ticket.limit_bound_ceiling_for_test(Duration::from_millis(1000))?; let session = ticket.bound_session()?; let (release, entered) = p.pause_for_test().await; let observer = ticket.publish(&p, ready(&session).await?)?; @@ -484,8 +507,12 @@ async fn bound_final_queued_transport_rechecks_ceiling_before_initial_execution( async fn bound_final_observes_shared_coordinator_recovery_without_losing_lifecycle_admission() -> Result { let f = Fixture::new(ObjectFormat::Sha256).await?; - let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; - let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let p = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; let ticket = super::bound::bind(&f, &c, [237; 16], "owner").await?; let session = ticket.bound_session()?; p.fault_for_test(2); diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service/restore.rs b/crates/canopy-server/src/packs/publication/tests/staging_service/restore.rs new file mode 100644 index 00000000..f85628af --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/staging_service/restore.rs @@ -0,0 +1,490 @@ +//! Real durable command/owner recovery; no old coordinator or usable session. +use super::*; +use cellule_runtime::{Committed, PendingMutation, Resolution}; + +async fn execute(f: &Fixture, action: CustodyAction) -> Result> { + let ready = PreparedCustody::prepare(&f.client(), &f.target, action, identity()?).await?; + Ok(ready + .register(&f.client(), identity()?) + .await? + .recover(&f.client()) + .await?) +} +fn granted_token(value: &CustodyReply) -> Result { + match value { + CustodyReply::Preparation(PreparationReply::Granted(lease)) => Ok(lease.token), + CustodyReply::Staging(StagingReply::Granted(lease)) => Ok(lease.token), + _ => Err("custody grant missing".into()), + } +} +async fn head( + f: &Fixture, + kind: u8, + execute_original: bool, +) -> Result<(PendingMutation, Option>)> { + head_expiring(f, kind, execute_original, execute_original).await +} +pub(in crate::packs::publication::tests) async fn head_expiring( + f: &Fixture, + kind: u8, + execute_original: bool, + short_expiry: bool, +) -> Result<(PendingMutation, Option>)> { + let input = f.begin([230 + kind; 16]); + let action = match kind { + 0 => CustodyAction::BeginStaging(input), + 4 => CustodyAction::BeginPreparation(input), + _ => { + let source = if kind < 4 { + CustodyAction::BeginStaging(input) + } else { + CustodyAction::BeginPreparation(input) + }; + let previous = granted_token(&execute(f, source).await?.output)?; + let lease = LeaseRequest { + check: check(previous), + lease_ms: DEFAULT_LEASE_MS, + }; + match kind { + 1 => CustodyAction::ClaimStaging(lease), + 2 => CustodyAction::RenewStaging(lease), + 3 => CustodyAction::BindStaging(lease.check), + 5 => CustodyAction::ClaimPreparation(lease), + 6 => CustodyAction::RenewPreparation(lease), + _ => return Err("unknown fixture kind".into()), + } + } + }; + let mut mutation = identity()?; + if short_expiry { + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + } + let command = PreparedCustody::prepare(&f.client(), &f.target, action, mutation).await?; + let original = command.evidence().clone(); + let registered = command.register(&f.client(), identity()?).await?; + let committed = if execute_original { + Some(registered.recover(&f.client()).await?) + } else { + None + }; + Ok((original, committed)) +} +async fn restore( + f: &Fixture, + client: CellClient, + kind: u8, + limits: StagingLimits, +) -> Result<(StagingCoordinator, StagingTicket)> { + let service = StagingCoordinator::new(f.target.clone(), limits, f.authority())?; + let ready = ReadyStaging::restore(client, f.target.clone(), [230 + kind; 16]).await?; + let ticket = service.submit(ready).map_err(|(error, _)| error)?; + Ok((service, ticket)) +} +async fn settle(ticket: &StagingTicket) -> Result { + Ok(timeout(Duration::from_secs(10), ticket.wait()).await?) +} +async fn expired(evidence: &PendingMutation) -> Result { + let until = evidence.identity().expires_at_ms; + let now = i64::try_from(SystemTime::now().duration_since(UNIX_EPOCH)?.as_millis())?; + if now <= until { + tokio::time::sleep(Duration::from_millis((until - now + 1) as u64)).await; + } + Ok(()) +} + +#[tokio::test] +async fn cold_staging_reconstructs_all_seven_heads_without_replacing_originals_or_clocks() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in 0..7 { + let f = Fixture::new(format).await?; + let (evidence, expected) = head(&f, kind, true).await?; + let expected = expected.ok_or("original not executed")?; + // A changed restart profile must not hide original knowledge or + // rewrite its command duration. New renewals use the new profile. + let limits = StagingLimits { + lease_ms: 10_000, + renew_before_ms: 5_000, + ..StagingLimits::default() + }; + let (service, ticket) = restore(&f, f.client(), kind, limits).await?; + let state = settle(&ticket).await?; + if kind < 3 { + assert!( + matches!(state, StagingState::Active(_)), + "kind {kind}: {state:?}" + ); + } else { + assert!( + matches!(state, StagingState::Bound(_)), + "kind {kind}: {state:?}" + ); + assert_eq!( + ticket.bound_session()?.lease.token, + granted_token(&expected.output)? + ); + } + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + assert_eq!( + *ticket.restored_outcome().ok_or("original reply lost")?, + expected + ); + let saved = RegisteredCustody::load_latest(&f.client(), &f.target, [230 + kind; 16]) + .await? + .ok_or("head missing")?; + assert_eq!(saved.evidence(), &evidence); + assert!(service.close_and_drain().await.is_empty()); + assert!(ticket.spawn(|_| async { Ok(()) }).is_err()); + assert_eq!( + *ticket.restored_outcome().ok_or("closed history lost")?, + expected + ); + assert_eq!(service.stats().admitted, 0); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn cold_staging_keeps_all_original_receipts_after_sdk_expiry_and_actual_owner_restore() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in 0..7 { + let f = Fixture::new(format).await?; + let (evidence, expected) = head(&f, kind, true).await?; + let expected = expected.ok_or("original not executed")?; + let old = granted_token(&expected.output)?; + let (runtime, handle, client) = + super::super::durable_recovery::restore_owner(&f, &check(old)).await?; + assert_ne!(handle.owner_fence(), old.owner); + expired(&evidence).await?; + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Expired + )); + let (service, ticket) = + restore(&f, client.clone(), kind, StagingLimits::default()).await?; + assert!(matches!(settle(&ticket).await?, StagingState::Fenced(_))); + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + assert_eq!( + *ticket.restored_outcome().ok_or("old-owner history lost")?, + expected + ); + assert!(ticket.bound_session().is_err()); + assert!(ticket.spawn(|_| async { Ok(()) }).is_err()); + assert!(service.close_and_drain().await.is_empty()); + let saved = RegisteredCustody::load_latest(&client, &f.target, [230 + kind; 16]) + .await? + .ok_or("old head missing")?; + assert_eq!(saved.evidence(), &evidence); + assert_eq!(saved.recover(&client).await?, expected); + runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn cold_staging_absent_originals_execute_or_fence_under_actual_new_owner_without_new_identity() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in 0..7 { + let f = Fixture::new(format).await?; + let (evidence, expected) = head(&f, kind, false).await?; + assert!(expected.is_none()); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + let (runtime, handle, client) = + super::super::durable_recovery::restore_owner_fence(&f, f.handle.owner_fence()) + .await?; + let (service, ticket) = + restore(&f, client.clone(), kind, StagingLimits::default()).await?; + let state = settle(&ticket).await?; + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + let saved = RegisteredCustody::load_latest(&client, &f.target, [230 + kind; 16]) + .await? + .ok_or("registered original lost")?; + assert_eq!(saved.evidence(), &evidence); + if matches!(kind, 2 | 3 | 6) { + assert!( + matches!(state, StagingState::Fenced(_)), + "old-token kind {kind}: {state:?}" + ); + let outcome = ticket.restored_outcome().ok_or("stale denial lost")?; + assert!( + matches!( + &outcome.output, + CustodyReply::Preparation(PreparationReply::Denied( + PreparationDenial::Stale + )) | CustodyReply::Staging(StagingReply::Denied(PreparationDenial::Stale)) + ), + "kind {kind}: {outcome:?}" + ); + assert!(saved.settled()); + assert!( + matches!(saved.recover(&client).await, Err(InvocationError::Rejected(value)) if *value == *outcome) + ); + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Committed(_) + )); + } else { + let outcome = ticket + .restored_outcome() + .ok_or("absent original reply lost")?; + assert_eq!(granted_token(&outcome.output)?.owner, handle.owner_fence()); + assert_eq!(saved.recover(&client).await?, *outcome); + assert!( + matches!(state, StagingState::Active(_) | StagingState::Bound(_)), + "kind {kind}: {state:?}" + ); + } + assert!(service.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn cold_staging_denials_remain_original_after_authority_is_repaired() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in [0, 4] { + let f = Fixture::new(format).await?; + let action = if kind == 0 { + CustodyAction::BeginStaging(f.begin([230 + kind; 16])) + } else { + CustodyAction::BeginPreparation(f.begin([230 + kind; 16])) + }; + let prepared = + PreparedCustody::prepare(&f.client(), &f.target, action, identity()?).await?; + let evidence = prepared.evidence().clone(); + let registered = prepared.register(&f.client(), identity()?).await?; + super::super::publishing::edit(&f, "UPDATE repository_identity SET owner='other'") + .await?; + let expected = match registered.recover(&f.client()).await { + Err(InvocationError::Rejected(value)) => *value, + value => return Err(format!("expected original denial: {value:?}").into()), + }; + super::super::publishing::edit(&f, "UPDATE repository_identity SET owner='owner'") + .await?; + let (service, ticket) = restore(&f, f.client(), kind, StagingLimits::default()).await?; + assert!(matches!(settle(&ticket).await?, StagingState::Fenced(_))); + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + assert_eq!( + *ticket.restored_outcome().ok_or("original denial missing")?, + expected + ); + assert!(ticket.spawn(|_| async { Ok(()) }).is_err()); + assert!(ticket.bound_session().is_err()); + assert_eq!(f.counts().await?, (0, 0)); + assert!(service.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn cold_staging_reply_loss_and_query_failures_retain_originals_through_closed_recovery() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in [0, 3, 6] { + for fault in 0..4 { + let f = Fixture::new(format).await?; + let (evidence, _) = head(&f, kind, false).await?; + let ready = + ReadyStaging::restore(f.client(), f.target.clone(), [230 + kind; 16]).await?; + let service = StagingCoordinator::new( + f.target.clone(), + StagingLimits::default(), + f.authority(), + )?; + if fault == 0 { + // The ready capability has the authentic original bytes, + // but a failed phase read still cannot authorize execution. + super::super::publishing::edit( + &f, + "ALTER TABLE catalog_custody_commands RENAME TO custody_query_fault", + ) + .await?; + } else { + service.fault_for_test(fault); + } + let ticket = service.submit(ready).map_err(|(error, _)| error)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + assert!(ticket.restored_outcome().is_none()); + assert_eq!( + service.stats().command_bytes, + super::super::super::custody::RESERVATION + ); + drop(ticket); + let ticket = service + .pending([230 + kind; 16]) + .ok_or("dropped observer lost original")?; + assert_eq!(service.close_and_drain().await.len(), 1); + if fault == 0 { + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + super::super::publishing::edit( + &f, + "ALTER TABLE custody_query_fault RENAME TO catalog_custody_commands", + ) + .await?; + } + service.recover(&ticket)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Stopped | StagingState::Bound(_) | StagingState::Fenced(_) + )); + let original = ticket + .restored_outcome() + .ok_or("closed recovery lost original reply")?; + let saved = + RegisteredCustody::load_latest(&f.client(), &f.target, [230 + kind; 16]) + .await? + .ok_or("original disappeared")?; + assert_eq!(saved.evidence(), &evidence); + assert_eq!(saved.recover(&f.client()).await?, *original); + assert!(ticket.bound_session().is_err()); + assert!(ticket.spawn(|_| async { Ok(()) }).is_err()); + assert!(service.close_and_drain().await.is_empty()); + assert_eq!(service.stats().command_bytes, 0); + f.runtime.shutdown().await?; + } + } + } + Ok(()) +} + +#[tokio::test] +async fn cold_bound_owner_loss_cancels_workers_before_releasing_resource_credit() -> Result { + use std::sync::atomic::{AtomicBool, Ordering}; + struct Resource { + service: StagingCoordinator, + dropped: Arc, + wrong_order: Arc, + } + impl Drop for Resource { + fn drop(&mut self) { + if self.service.stats().workers != 1 { + self.wrong_order.store(true, Ordering::Release); + } + self.dropped.store(true, Ordering::Release); + } + } + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in [3, 4, 6] { + let f = Fixture::new(format).await?; + let (evidence, expected) = head(&f, kind, true).await?; + let expected = expected.ok_or("original missing")?; + let (service, ticket) = restore(&f, f.client(), kind, StagingLimits::default()).await?; + assert!(matches!(settle(&ticket).await?, StagingState::Bound(_))); + let session = ticket.bound_session()?; + let dropped = Arc::new(AtomicBool::new(false)); + let wrong_order = Arc::new(AtomicBool::new(false)); + let resource = Resource { + service: service.clone(), + dropped: dropped.clone(), + wrong_order: wrong_order.clone(), + }; + let (entered, running) = oneshot::channel(); + let worker = ticket.spawn_bound(move |_, _context| async move { + let _resource = resource; + let _ = entered.send(()); + std::future::pending::>().await + })?; + timeout(Duration::from_secs(10), running).await??; + let (runtime, handle, _) = + super::super::durable_recovery::restore_owner(&f, &session.check).await?; + assert_ne!(handle.owner_fence(), session.lease.token.owner); + assert!(session.check_owner().await.is_err()); + // A clone observing the fence after it fired must also wake. + timeout(Duration::from_secs(2), session.clone().wait_fenced()).await?; + // No manual renewal, clock advance, coordinator stop or job-state + // mutation: the session's permanent shared fence must wake work. + assert!( + timeout(Duration::from_secs(2), worker.wait()) + .await? + .is_err() + ); + assert!(session.live_lease().is_err()); + assert!(dropped.load(Ordering::Acquire)); + assert!(!wrong_order.load(Ordering::Acquire)); + assert_eq!(service.stats().workers, 0); + assert!(ticket.spawn_bound(|_, _context| async { Ok(()) }).is_err()); + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + assert_eq!( + *ticket.restored_outcome().ok_or("historical outcome lost")?, + expected + ); + assert!(service.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn cold_staging_unsettled_expired_originals_keep_exact_evidence_and_never_execute() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in 0..7 { + let f = Fixture::new(format).await?; + let (evidence, expected) = head_expiring(&f, kind, false, true).await?; + assert!(expected.is_none()); + let (runtime, _, client) = + super::super::durable_recovery::restore_owner_fence(&f, f.handle.owner_fence()) + .await?; + expired(&evidence).await?; + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Expired + )); + let (service, ticket) = + restore(&f, client.clone(), kind, StagingLimits::default()).await?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + assert_eq!(ticket.restored_evidence(), Some(&evidence)); + assert!(ticket.restored_outcome().is_none()); + assert!(ticket.spawn(|_| async { Ok(()) }).is_err()); + assert!(ticket.spawn_bound(|_, _context| async { Ok(()) }).is_err()); + assert_eq!( + service.stats().command_bytes, + super::super::super::custody::RESERVATION + ); + assert_eq!(service.close_and_drain().await.len(), 1); + service.recover(&ticket)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + let saved = RegisteredCustody::load_latest(&client, &f.target, [230 + kind; 16]) + .await? + .ok_or("expired head lost")?; + assert_eq!(saved.evidence(), &evidence); + assert!(!saved.settled()); + assert!( + matches!(saved.recover(&client).await, Err(InvocationError::Pending(value)) if *value == evidence) + ); + assert!(matches!( + client.resolve(&evidence).await?, + Resolution::Expired + )); + assert_eq!(service.close_and_drain().await.len(), 1); + runtime.shutdown().await?; + } + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service/retirement.rs b/crates/canopy-server/src/packs/publication/tests/staging_service/retirement.rs new file mode 100644 index 00000000..13022519 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/staging_service/retirement.rs @@ -0,0 +1,508 @@ +//! Automatic closure must preserve original knowledge and resource ownership. +use super::super::publishing::edit; +use super::*; +use cellule_runtime::{PendingMutation, Resolution}; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +async fn expired(evidence: &PendingMutation) -> Result { + let now = crate::packs::publication::sql::now(0)?; + if now <= evidence.identity().expires_at_ms { + tokio::time::sleep(Duration::from_millis( + (evidence.identity().expires_at_ms - now + 1) as u64, + )) + .await; + } + Ok(()) +} +async fn until(c: &StagingCoordinator, predicate: impl Fn(StagingStats) -> bool) -> Result { + timeout(Duration::from_secs(10), async { + while !predicate(c.stats()) { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + Ok(()) +} +async fn head(f: &Fixture, operation: [u8; 16]) -> Result { + Ok( + RegisteredCustody::load_latest(&f.client(), &f.target, operation) + .await? + .ok_or("custody head missing")?, + ) +} +async fn stop(f: &Fixture, original: &RegisteredCustody) -> Result { + let ready = original + .ready_stop(f.client(), identity()?, &f.authority()) + .await?; + assert_eq!( + ready.command_for_test().execute().await?.output, + CustodyStopReply::Stopped + ); + Ok(()) +} + +#[tokio::test] +async fn automatic_retirement_closes_all_seven_cold_originals_after_observer_drop_and_service_close() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in 0..7 { + let f = Fixture::new(format).await?; + let (evidence, _) = restore::head_expiring(&f, kind, false, true).await?; + expired(&evidence).await?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let ticket = c + .submit( + ReadyStaging::restore(f.client(), f.target.clone(), [230 + kind; 16]).await?, + ) + .map_err(|(error, _)| error)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + assert_eq!(c.close_and_drain().await.len(), 1); + drop(ticket); + let original = head(&f, [230 + kind; 16]).await?; + let counts = f.counts().await?; + stop(&f, &original).await?; + until(&c, |s| s.admitted == 0 && !s.retirement_running).await?; + assert!(c.pending([230 + kind; 16]).is_none()); + assert_eq!(c.stats().command_bytes, 0); + assert_eq!(c.stats().retirement_recoveries, 1); + assert_eq!(f.counts().await?, counts); + let saved = head(&f, [230 + kind; 16]).await?; + assert_eq!(saved.evidence(), &evidence); + assert!(saved.closed()); + assert!(!saved.settled()); + assert!(matches!(saved.recover(&f.client()).await, + Err(InvocationError::Pending(value)) if *value == evidence)); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +struct Owned { + coordinator: StagingCoordinator, + dropped: Arc, + wrong: Arc, +} +impl Drop for Owned { + fn drop(&mut self) { + let stats = self.coordinator.stats(); + if stats.workers == 0 || stats.admitted == 0 || stats.command_bytes == 0 { + self.wrong.store(true, Ordering::Release); + } + self.dropped.fetch_add(1, Ordering::AcqRel); + } +} + +#[tokio::test] +async fn automatic_retirement_joins_live_callbacks_and_drops_retained_results_before_releasing_credits() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for bound in [false, true] { + let f = Fixture::new(format).await?; + let c = + StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let operation = [220; 16]; + let ticket = if bound { + bound::bind(&f, &c, operation, "owner").await? + } else { + let t = submit(&f, &c, operation, "owner").await?; + active(&t).await?; + t + }; + let session = if bound { + Some(ticket.bound_session()?) + } else { + None + }; + let dropped = Arc::new(AtomicUsize::new(0)); + let wrong = Arc::new(AtomicBool::new(false)); + let owned = || Owned { + coordinator: c.clone(), + dropped: dropped.clone(), + wrong: wrong.clone(), + }; + let finished = owned(); + let completed = if bound { + ticket.spawn_bound(move |_, _context| async move { Ok(finished) })? + } else { + ticket.spawn(move |_| async move { Ok(finished) })? + }; + let running = owned(); + let (entered, start) = oneshot::channel(); + let worker = if bound { + ticket.spawn_bound(move |_, _context| async move { + let _ = entered.send(()); + std::future::pending::<()>().await; + drop(running); + Ok(()) + })? + } else { + ticket.spawn(move |_| async move { + let _ = entered.send(()); + std::future::pending::<()>().await; + drop(running); + Ok(()) + })? + }; + timeout(Duration::from_secs(10), start).await??; + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + c.fault_for_test(1); // Registered original, execution never started. + ticket.renew_with_identity_for_test(mutation).await?; + until(&c, |s| s.uncertain == 1).await?; + assert!(matches!(ticket.state(), StagingState::Uncertain(_))); + let original = head(&f, operation).await?; + let evidence = original.evidence().clone(); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + expired(&evidence).await?; + assert_eq!(c.stats().workers, 2); // Includes the untransferred completed result. + assert_eq!(dropped.load(Ordering::Acquire), 0); + if let Some(session) = &session { + assert!(session.live_lease().is_ok()); + } + drop(completed); + drop(worker); + drop(ticket); + stop(&f, &original).await?; + until(&c, |s| s.admitted == 0 && !s.retirement_running).await?; + assert_eq!(dropped.load(Ordering::Acquire), 2); + assert!(!wrong.load(Ordering::Acquire)); + assert_eq!(c.stats().workers, 0); + assert_eq!(c.stats().command_bytes, 0); + assert_eq!(c.stats().retirement_recoveries, 1); + if let Some(session) = session { + assert!(session.live_lease().is_err()); + } + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + assert!(!head(&f, operation).await?.settled()); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn automatic_retirement_does_not_execute_absent_commands_or_retry_known_phases_without_a_stop() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (evidence, _) = restore::head_expiring(&f, 0, false, false).await?; + let ready = ReadyStaging::restore(f.client(), f.target.clone(), [230; 16]).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + edit( + &f, + "ALTER TABLE catalog_custody_commands RENAME TO private_query_unavailable", + ) + .await?; + let ticket = c.submit(ready).map_err(|(error, _)| error)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + until(&c, |s| s.retirement_failures >= 2).await?; + assert_eq!(f.counts().await?, (0, 0)); + assert_eq!(c.stats().admitted, 1); + assert_eq!(c.stats().retirement_recoveries, 0); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Absent + )); + edit( + &f, + "ALTER TABLE private_query_unavailable RENAME TO catalog_custody_commands", + ) + .await?; + let known = head(&f, [230; 16]).await?.recover(&f.client()).await?; + let probes = c.stats().retirement_probes; + until(&c, |s| s.retirement_probes >= probes + 2).await?; + assert!(matches!(ticket.state(), StagingState::Uncertain(_))); + assert_eq!(c.stats().retirement_recoveries, 0); + assert!(ticket.restored_outcome().is_none()); + assert_eq!(c.close_and_drain().await.len(), 1); + c.recover(&ticket)?; + until(&c, |s| s.admitted == 0 && !s.retirement_running).await?; + assert_eq!( + *ticket.restored_outcome().ok_or("known outcome lost")?, + known + ); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn exact_retirement_probe_keeps_old_ordinal_after_successor_and_rejects_corrupt_facts() +-> Result { + use crate::packs::publication::custody::OwnedCustody; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (evidence, _) = restore::head_expiring(&f, 0, false, true).await?; + let probe = OwnedCustody::restore(&f.client(), &f.target, [230; 16]) + .await? + .stop_probe()?; + assert!(!probe.observed(&f.client()).await?); + expired(&evidence).await?; + let original = head(&f, [230; 16]).await?; + stop(&f, &original).await?; + let successor = + PreparedCustody::prepare(&f.client(), &f.target, original.action()?, identity()?) + .await? + .register(&f.client(), identity()?) + .await?; + assert_ne!(head(&f, [230; 16]).await?.evidence(), &evidence); + assert_eq!(head(&f, [230; 16]).await?.evidence(), successor.evidence()); + assert!(probe.observed(&f.client()).await?); + let fact = f + .handle + .query(0, 4096, |db| { + Ok(db.query_row( + "SELECT stopped FROM catalog_custody_commands WHERE operation=?1 AND step=0", + [vec![230u8; 16]], + |row| row.get::<_, Vec>(0), + )?) + }) + .await?; + edit(&f, "DROP TRIGGER catalog_custody_stop_immutable").await?; + edit( + &f, + "UPDATE catalog_custody_commands SET stopped=x'01' WHERE step=0", + ) + .await?; + assert!(probe.observed(&f.client()).await.is_err()); + edit( + &f, + &format!( + "UPDATE catalog_custody_commands SET stopped=x'{}' WHERE step=0", + hex::encode(fact) + ), + ) + .await?; + assert!(probe.observed(&f.client()).await?); + assert!(matches!( + f.client().resolve(&evidence).await?, + Resolution::Expired + )); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn automatic_retirement_skips_bad_heads_and_revisits_them_after_repair() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (bad, _) = restore::head_expiring(&f, 0, false, true).await?; + let (good, _) = restore::head_expiring(&f, 4, false, true).await?; + expired(&bad).await?; + expired(&good).await?; + let bad_ready = ReadyStaging::restore(f.client(), f.target.clone(), [230; 16]).await?; + let good_ready = ReadyStaging::restore(f.client(), f.target.clone(), [234; 16]).await?; + let bad_original = head(&f, [230; 16]).await?; + let good_original = head(&f, [234; 16]).await?; + let intent = f + .handle + .query(0, 4096, |db| { + Ok(db.query_row( + "SELECT intent FROM catalog_custody_commands WHERE operation=?1 AND step=0", + [vec![230u8; 16]], + |row| row.get::<_, Vec>(0), + )?) + }) + .await?; + edit(&f, "DROP TRIGGER catalog_custody_identity_immutable").await?; + edit(&f, "UPDATE catalog_custody_commands SET intent=x'01' WHERE operation=x'e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6'").await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + let bad_ticket = c.submit(bad_ready).map_err(|(error, _)| error)?; + let good_ticket = c.submit(good_ready).map_err(|(error, _)| error)?; + assert!(matches!( + terminal(&bad_ticket).await?, + StagingState::Uncertain(_) + )); + assert!(matches!( + terminal(&good_ticket).await?, + StagingState::Uncertain(_) + )); + until(&c, |s| s.retirement_failures > 0).await?; + stop(&f, &good_original).await?; + until(&c, |s| s.admitted == 1 && s.retirement_recoveries == 1).await?; + assert!(matches!(good_ticket.state(), StagingState::Fenced(_))); + assert!(matches!(bad_ticket.state(), StagingState::Uncertain(_))); + assert_eq!( + c.stats().command_bytes, + crate::packs::publication::custody::RESERVATION + ); + edit(&f, &format!("UPDATE catalog_custody_commands SET intent=x'{}' WHERE operation=x'e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6'", hex::encode(intent))).await?; + stop(&f, &bad_original).await?; + until(&c, |s| s.admitted == 0 && !s.retirement_running).await?; + assert_eq!(c.stats().retirement_recoveries, 2); + assert!(bad_ticket.restored_outcome().is_none()); + assert!(good_ticket.restored_outcome().is_none()); + assert!(matches!( + f.client().resolve(&bad).await?, + Resolution::Expired + )); + assert!(matches!( + f.client().resolve(&good).await?, + Resolution::Expired + )); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn automatic_retirement_does_not_apply_old_custody_closure_to_an_input_checkpoint() -> Result +{ + use canopy_object_storage::artifact::ArtifactStore; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let (old, _) = restore::head_expiring(&f, 0, false, true).await?; + expired(&old).await?; + stop(&f, &head(&f, [230; 16]).await?).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default(), f.authority())?; + PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginStaging(f.begin([230; 16])), + identity()?, + ) + .await? + .register(&f.client(), identity()?) + .await?; + let ticket = c + .submit(ReadyStaging::restore(f.client(), f.target.clone(), [230; 16]).await?) + .map_err(|(e, _)| e)?; + active(&ticket).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let proof = super::super::inputs::seal(&f, &ticket, store, 2).await?; + c.fault_for_test(1); + let checkpoint = ticket + .register_inputs(proof, identity()?) + .map_err(|(e, _)| e)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + let (other, _) = restore::head_expiring(&f, 4, false, true).await?; + expired(&other).await?; + let other_ticket = c + .submit(ReadyStaging::restore(f.client(), f.target.clone(), [234; 16]).await?) + .map_err(|(e, _)| e)?; + assert!(matches!( + terminal(&other_ticket).await?, + StagingState::Uncertain(_) + )); + until(&c, |s| s.retirement_probes >= 2).await?; + stop(&f, &head(&f, [234; 16]).await?).await?; + until(&c, |s| s.admitted == 1 && !s.retirement_running).await?; + assert!(matches!(ticket.state(), StagingState::Uncertain(_))); + assert_eq!(c.stats().retirement_recoveries, 1); + assert_eq!( + c.stats().command_bytes, + crate::packs::publication::custody::RESERVATION + 4096 + ); + assert!(checkpoint.wait().await.is_err()); + c.recover(&ticket)?; + checkpoint.wait().await.map_err(|e| e.to_string())?; + assert!(c.close_and_drain().await.is_empty()); + assert!(matches!( + f.client().resolve(&old).await?, + Resolution::Expired + )); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn automatic_retirement_spans_more_than_one_page_and_restarts_at_earlier_keys_after_idle() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits { + operations: 256, + per_actor: 192, + ..StagingLimits::default() + }, + f.authority(), + )?; + let mut originals = Vec::new(); + for n in 1u16..=130 { + let mut operation = [0; 16]; + operation[14..].copy_from_slice(&n.to_be_bytes()); + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let saved = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginStaging(f.begin(operation)), + mutation, + ) + .await? + .register(&f.client(), identity()?) + .await?; + originals.push((operation, saved)); + } + expired(originals.last().ok_or("no originals")?.1.evidence()).await?; + for (operation, _) in &originals { + drop( + c.submit(ReadyStaging::restore(f.client(), f.target.clone(), *operation).await?) + .map_err(|(error, _)| error)?, + ); + } + until(&c, |s| s.uncertain == 130).await?; + for (_, saved) in &originals { + stop(&f, saved).await?; + } + until(&c, |s| s.admitted == 0 && !s.retirement_running).await?; + assert_eq!(c.stats().retirement_recoveries, 130); + assert_eq!(c.stats().command_bytes, 0); + assert_eq!(f.counts().await?, (0, 0)); + // A new unknown original below the previous cursor starts a fresh singleton. + let mut mutation = identity()?; + mutation.expires_at_ms = mutation.issued_at_ms + 1_000; + let operation = originals[0].0; + let saved = PreparedCustody::prepare( + &f.client(), + &f.target, + CustodyAction::BeginStaging(f.begin(operation)), + mutation, + ) + .await? + .register(&f.client(), identity()?) + .await?; + expired(saved.evidence()).await?; + let ticket = c + .submit(ReadyStaging::restore(f.client(), f.target.clone(), operation).await?) + .map_err(|(error, _)| error)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + drop(ticket); + stop(&f, &saved).await?; + until(&c, |s| s.admitted == 0 && !s.retirement_running).await?; + assert_eq!(c.stats().retirement_recoveries, 131); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/terminal_retention.rs b/crates/canopy-server/src/packs/publication/tests/terminal_retention.rs index 55accde2..3a1690ce 100644 --- a/crates/canopy-server/src/packs/publication/tests/terminal_retention.rs +++ b/crates/canopy-server/src/packs/publication/tests/terminal_retention.rs @@ -5,6 +5,16 @@ use canopy_object_storage::artifact::ArtifactStore; use cellule_runtime::{CellClient, Committed, Resolution}; use tokio::time::{Duration, timeout}; +async fn unrelated_recovery_pins(handle: &CellHandle, token: PreparationToken) -> Result> { + Ok(handle.query(0, 64 << 10, move |db| { + let mut statement = db.prepare("SELECT incarnation,admission_sequence,recovery,recovery_phase,recovery_phase_revision FROM catalog_leases WHERE recovery IS NOT NULL AND NOT(incarnation=?1 AND admission_sequence=?2) ORDER BY incarnation,admission_sequence")?; + let pins = statement.query_map(rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |row| Ok(( + row.get::<_, Vec>(0)?, row.get::<_, u64>(1)?, row.get::<_, Vec>(2)?, row.get::<_, Option>>(3)?, row.get::<_, u64>(4)? + )))?.collect::>>()?; + serde_json::to_vec(&pins).map_err(|_| Error::Command("fixture unrelated recovery pins")) + }).await?) +} + pub(super) async fn maintenance( handle: &CellHandle, repository: [u8; 16], @@ -77,7 +87,7 @@ pub(super) async fn qualify(context: Context<'_>, fault: u8, provider: Arc, fault: u8, provider: Arc, fault: u8, provider: Arc, fault: u8, provider: Arc, fault: u8, ) -> Result { + let token = head.token(); + let unrelated = unrelated_recovery_pins(handle, token).await?; let admin = maintenance(handle, f.repository).await?; // The certificate binds the original actor, but release needs CURRENT // repository administration and actual admitted owner custody separately. @@ -349,19 +366,19 @@ pub(super) async fn archive( Resolution::Absent )); handle - .query(0, 128, |db| { + .query(0, 128, move |db| { assert_eq!( db.query_row( - "SELECT count(*) FROM pushes WHERE recovery IS NOT NULL", - [], + "SELECT count(*) FROM catalog_recovery_receipts WHERE incarnation=?1 AND admission_sequence=?2", + rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |r| r.get::<_, u64>(0) )?, 0 ); assert_eq!( db.query_row( - "SELECT count(*) FROM catalog_leases WHERE recovery IS NOT NULL", - [], + "SELECT count(*) FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND recovery IS NOT NULL", + rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |r| r.get::<_, u64>(0) )?, 1 @@ -370,7 +387,11 @@ pub(super) async fn archive( }) .await?; edit_handle(handle, "DROP TRIGGER terminal_release_late_fault").await?; - let queue = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let queue = PublicationCoordinator::new( + f.target.clone(), + PublicationLimits::default(), + f.publication_budget.clone(), + )?; queue.fault_for_test(if fault == 4 { 2 } else { fault }); let limits = RecoveryScanLimits { page: 1, @@ -383,7 +404,8 @@ pub(super) async fn archive( f.target.clone(), store.clone(), queue.clone(), - limits, + f.scans(limits), + f.authority(), admin.clone(), )?; timeout(Duration::from_secs(10), entered).await??; @@ -423,7 +445,8 @@ pub(super) async fn archive( f.target.clone(), store.clone(), queue.clone(), - limits, + f.scans(limits), + f.authority(), admin.clone(), )?; service.shutdown().await?; @@ -434,7 +457,8 @@ pub(super) async fn archive( f.target.clone(), store.clone(), queue.clone(), - limits, + f.scans(limits), + f.authority(), admin.clone(), )?; // No caller recovery request: the supervisor retries the original @@ -449,7 +473,10 @@ pub(super) async fn archive( assert!(stats.release_recovered > 0); assert_eq!(stats.failures, 0); if fault != 1 { - assert_eq!(stats.scanned, 0); + // A retired push has no pin. Other settled intermediate heads + // must not admit another publication or release. + assert_eq!(stats.scanned, stats.settled); + assert_eq!(stats.submitted, 0); } } let PublicationState::Finished(Ok(PublicationOutcome::TerminalRelease(released))) = @@ -469,14 +496,15 @@ pub(super) async fn archive( (&recovered.output, recovered.receipt), (&released.output, released.receipt) ); - handle.query(0, 128, |db| { - assert_eq!(db.query_row("SELECT count(*) FROM catalog_leases WHERE recovery IS NOT NULL", [], |r| r.get::<_, u64>(0))?, 0); - assert_eq!(db.query_row("SELECT count(*) FROM pushes WHERE recovery IS NOT NULL AND recovery_phase IS NOT NULL AND recovery_release IS NOT NULL", [], |r| r.get::<_, u64>(0))?, 1); - for sql in ["UPDATE pushes SET recovery=NULL,recovery_phase=NULL,recovery_release=NULL WHERE recovery IS NOT NULL", "UPDATE pushes SET recovery_phase=x'01' WHERE recovery IS NOT NULL", "UPDATE pushes SET recovery_release=x'01' WHERE recovery IS NOT NULL", "DELETE FROM pushes WHERE recovery IS NOT NULL"] { + handle.query(0, 128, move |db| { + assert_eq!(db.query_row("SELECT count(*) FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND recovery IS NOT NULL", rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |r| r.get::<_, u64>(0))?, 0); + assert_eq!(db.query_row("SELECT count(*) FROM catalog_recovery_receipts WHERE incarnation=?1 AND admission_sequence=?2", rusqlite::params![token.owner.incarnation.as_bytes().as_slice(), token.attempt], |r| r.get::<_, u64>(0))?, 1); + for sql in ["UPDATE catalog_recovery_receipts SET recovery=NULL,recovery_phase=NULL,recovery_release=NULL WHERE recovery IS NOT NULL", "UPDATE catalog_recovery_receipts SET recovery_phase=x'01' WHERE recovery IS NOT NULL", "UPDATE catalog_recovery_receipts SET recovery_release=x'01' WHERE recovery IS NOT NULL", "DELETE FROM catalog_recovery_receipts WHERE recovery IS NOT NULL"] { assert!(db.execute(sql, []).is_err(), "{sql}"); } Ok(Vec::new()) }).await?; + assert_eq!(unrelated_recovery_pins(handle, token).await?, unrelated); let loaded = RegisteredRootRecovery::load( client, &f.target, @@ -491,11 +519,13 @@ pub(super) async fn archive( let original = loaded.clone(); let artifacts = store.clone(); let reader = client.clone(); + let authority = f.authority(); let actual = tokio::spawn(async move { original .dispatch_any( &reader, &artifacts, + &authority, &std::sync::atomic::AtomicBool::new(false), ) .await diff --git a/crates/canopy-server/src/packs/ref_state/mod.rs b/crates/canopy-server/src/packs/ref_state/mod.rs index d3ad6c6a..51d88879 100644 --- a/crates/canopy-server/src/packs/ref_state/mod.rs +++ b/crates/canopy-server/src/packs/ref_state/mod.rs @@ -1,6 +1,7 @@ //! Conditional immutable ref state; raw roots do not confer publication rights. //! Final owner/ACL/policy/root-CAS and durable outcome publication remain Cell -//! responsibilities. This data plane is not selected by the serving path yet. +//! responsibilities. Certified serving snapshots select this data plane; raw +//! roots and standalone index clients confer neither Read nor retention authority. use super::directory::index::{IndexError, NodeRef, RangeCursor, RangeIndex, ReadStats}; use crate::{ObjectFormat, PushPlan, RefExpectation}; use canopy_object_storage::artifact::ArtifactStore; diff --git a/crates/canopy-server/src/packs/sources/tests.rs b/crates/canopy-server/src/packs/sources/tests.rs index c7c18785..b537df90 100644 --- a/crates/canopy-server/src/packs/sources/tests.rs +++ b/crates/canopy-server/src/packs/sources/tests.rs @@ -3,6 +3,7 @@ use super::super::metadata::{ tests::{builder, fill, fixture}, }; use super::*; +mod changes; use canopy_object_storage::artifact::ArtifactStore; use canopy_object_storage::external::MAX_ARTIFACT_BYTES; use cellule_ltx::DiskBudget; diff --git a/crates/canopy-server/src/packs/sources/tests/changes.rs b/crates/canopy-server/src/packs/sources/tests/changes.rs new file mode 100644 index 00000000..3337fa66 --- /dev/null +++ b/crates/canopy-server/src/packs/sources/tests/changes.rs @@ -0,0 +1,184 @@ +//! Coverage transferred from the retired SQL hydration sequence to the actual +//! source descriptor cursor used by native write-base construction. +use super::*; +use crate::packs::directory::index::IndexRecord; +use cellule_runtime::codec::BoundedEncoder; + +async fn collect( + index: &SourceIndex, + before: Option, + after: Option, +) -> Result> { + let mut cursor = index.changes(before, after, None)?; + let mut records = Vec::new(); + loop { + let page = cursor.page(7, 64 << 10).await?; + if page.is_empty() { + break; + } + records.extend(page); + } + Ok(records) +} + +#[tokio::test] +async fn lower_keys_are_found_and_later_publications_are_excluded() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (store, _) = store(); + let index = SourceIndex::new(store, format); + let before = index + .build_sorted([5; 16], (200..460).map(|n| Ok(source(n, format)))) + .await?; + let selected = Some(index.insert(before, [6; 16], source(1, format)).await?); + let future = Some(index.insert(selected, [7; 16], source(0, format)).await?); + assert_eq!( + collect(&index, before, selected).await?, + [source(1, format)] + ); + assert_eq!( + collect(&index, selected, future).await?, + [source(0, format)] + ); + assert_eq!(collect(&index, selected, selected).await?, []); + assert_eq!(collect(&index, future, before).await?, []); // removals need no new native input + } + Ok(()) +} + +#[tokio::test] +async fn byte_limited_pages_and_restart_advance_only_over_returned_prefix() -> Result { + let format = ObjectFormat::Sha256; + let (store, _) = store(); + let index = SourceIndex::new(store, format); + let root = index + .build_sorted([5; 16], (0..270).map(|n| Ok(source(n, format)))) + .await?; + let mut encoder = BoundedEncoder::new(64 << 10)?; + source(0, format).encode_record(&mut encoder)?; + let size = encoder.finish().len(); + let mut cursor = index.changes(None, root, None)?; + let mut returned = Vec::new(); + for n in 0..270 { + let page = cursor.page(128, size * 2 - 1).await?; + assert_eq!(page, [source(n, format)]); + // A stateless restart cannot skip the prefetched but excluded descriptor. + let mut retry = index.changes(None, root, Some(page[0].key()))?; + let next = retry.page(1, size).await?; + assert_eq!( + next, + if n == 269 { + vec![] + } else { + vec![source(n + 1, format)] + } + ); + returned.extend(page); + } + assert!(cursor.page(128, size).await?.is_empty()); + assert_eq!(returned.len(), 270); + let mut cursor = index.changes(None, root, None)?; + assert!(matches!( + cursor.page(1, size - 1).await, + Err(IndexError::Limit) + )); + assert!(matches!( + cursor.page(1, size).await, + Err(IndexError::Integrity) + )); + Ok(()) +} + +#[tokio::test] +async fn duplicate_failed_update_deletion_and_replacement_preserve_changes() -> Result { + let format = ObjectFormat::Sha1; + let (store, _) = store(); + let index = SourceIndex::new(store, format); + let before = Some(index.insert(None, [5; 16], source(10, format)).await?); + assert_eq!( + Some(index.insert(before, [6; 16], source(10, format)).await?), + before + ); + let mut changed = source(10, format); + changed.index.manifest_digest[0] ^= 1; + assert!(index.insert(before, [6; 16], changed).await.is_err()); + assert!(collect(&index, before, before).await?.is_empty()); + let removed = index.remove(before, [7; 16], source(10, format)).await?; + let after = Some(index.insert(removed, [8; 16], source(1, format)).await?); + assert_eq!(collect(&index, before, after).await?, [source(1, format)]); + let replaced = Some( + index + .replace(before.ok_or("root")?, [9; 16], source(10, format), changed) + .await?, + ); + assert_eq!(collect(&index, before, replaced).await?, [changed]); + assert_eq!( + index.find(before, source(10, format).key()).await?, + Some(source(10, format)) + ); + Ok(()) +} + +#[tokio::test] +async fn small_increment_skips_unchanged_subtrees_after_ten_thousand_sources() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (store, _) = store(); + let index = SourceIndex::new(store, format); + let before = index + .build_sorted([5; 16], (100..10_100).map(|n| Ok(source(n, format)))) + .await?; + let mut after = before; + for n in [0, 50, 20_000] { + after = Some(index.insert(after, [6; 16], source(n, format)).await?); + } + index.clear_cache()?; + let initial = index.stats(); + assert_eq!( + collect(&index, before, after).await?, + [ + source(0, format), + source(50, format), + source(20_000, format) + ] + ); + assert!(index.stats().loaded_nodes - initial.loaded_nodes < 24); + } + Ok(()) +} + +#[tokio::test] +async fn different_tree_shapes_and_rewrites_match_independent_key_difference() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (store, _) = store(); + let index = SourceIndex::new(store, format); + for count in [1, 127, 128, 129, 300] { + let old: Vec<_> = (0..count).map(|n| source(n * 3, format)).collect(); + let mut new: Vec<_> = old + .iter() + .enumerate() + .filter(|(n, _)| n % 5 != 0) + .map(|(n, r)| { + let mut r = *r; + if n % 7 == 0 { + r.index.manifest_digest[0] ^= 1; + } + r + }) + .collect(); + new.extend( + (0..count) + .filter(|n| n % 4 == 0) + .map(|n| source(n * 3 + 1, format)), + ); + new.sort_by_key(|r| r.key()); + let before = index + .build_sorted([5; 16], old.iter().copied().map(Ok)) + .await?; + let after = index + .build_sorted([6; 16], new.iter().copied().map(Ok)) + .await?; + let expected: Vec<_> = new.iter().copied().filter(|r| !old.contains(r)).collect(); + assert_eq!(collect(&index, before, after).await?, expected); + } + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/verification/mod.rs b/crates/canopy-server/src/packs/verification/mod.rs index 781a5544..c75ecb23 100644 --- a/crates/canopy-server/src/packs/verification/mod.rs +++ b/crates/canopy-server/src/packs/verification/mod.rs @@ -25,9 +25,17 @@ impl CanonicalVerifier { git_dir: &Path, format: ObjectFormat, native: &crate::native_resources::NativeScope, + ) -> Result { + Self::new_owned(git_dir, format, native, std::sync::Arc::new(())) + } + pub(crate) fn new_owned( + git_dir: &Path, + format: ObjectFormat, + native: &crate::native_resources::NativeScope, + owner: crate::git_objects::ReadOwner, ) -> Result { Ok(Self { - native: GitObjects::batch(git_dir, native)?, + native: GitObjects::batch_owned(git_dir, native, owner)?, format, }) } diff --git a/crates/canopy-server/src/packs/verification/physical.rs b/crates/canopy-server/src/packs/verification/physical.rs index c05ea7f5..f793b5b1 100644 --- a/crates/canopy-server/src/packs/verification/physical.rs +++ b/crates/canopy-server/src/packs/verification/physical.rs @@ -43,6 +43,8 @@ impl Default for PhysicalLimits { } #[derive(Debug, thiserror::Error)] pub enum PhysicalError { + #[error("physical verification has no live staging custody")] + Staging(#[from] crate::packs::publication::StagingError), #[error("isolated native workspace failed")] Cache(#[from] CacheError), #[error("physical artifact binding failed")] @@ -113,6 +115,7 @@ impl PhysicalPackWitness { /// The service must also retain its native-process admission for this lifetime. pub struct PhysicalVerifier { store: ArtifactStore, + context: Option, // Drop the native actor/index before releasing the fenced workspace. native: Option, binding: Arc, @@ -127,6 +130,7 @@ pub struct PhysicalVerifier { failed: bool, } impl PhysicalVerifier { + #[cfg(test)] pub async fn download( root: &Path, budget: DiskBudget, @@ -134,6 +138,62 @@ impl PhysicalVerifier { descriptor: NativePackDescriptor, limits: PhysicalLimits, native: crate::native_resources::NativeScope, + ) -> Result { + Self::download_owned( + root, + budget, + store, + descriptor, + limits, + native, + Arc::new(()), + ) + .await + } + + /// Retain the admitted creating or bound worker through provider reads, + /// blocking assembly, native descendants and deferred workspace cleanup. + /// A retained owner does not extend custody; check it before each new stage. + pub async fn download_staged( + context: &crate::packs::publication::StagingContext, + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + descriptor: NativePackDescriptor, + limits: PhysicalLimits, + native: crate::native_resources::NativeScope, + ) -> Result { + let token = context.token()?; + if descriptor.repository != token.repository + || store.repository() != token.repository + || descriptor.operation != token.artifact_operation + || descriptor.format != context.format() + { + return Err(PhysicalError::Integrity); + } + let mut verifier = Self::download_owned( + root, + budget, + store, + descriptor, + limits, + native, + context.physical_owner(), + ) + .await?; + context.ensure_live()?; + verifier.context = Some(context.clone()); + Ok(verifier) + } + + async fn download_owned( + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + descriptor: NativePackDescriptor, + limits: PhysicalLimits, + native: crate::native_resources::NativeScope, + owner: crate::git_objects::ReadOwner, ) -> Result { descriptor.validate(store.repository(), descriptor.format)?; if limits.max_pack_bytes > MAX_ARTIFACT_BYTES @@ -145,16 +205,28 @@ impl PhysicalVerifier { return Err(PhysicalError::Limit); } let root = root.to_owned(); - let root = tokio::task::spawn_blocking(move || std::fs::canonicalize(root)).await??; - let cache = GitCache::create( + let root_owner = owner.clone(); + let root = tokio::task::spawn_blocking(move || { + let _owner = root_owner; + std::fs::canonicalize(root) + }) + .await??; + let cache = GitCache::create_owned( root.clone(), budget.clone(), "refs/heads/main", descriptor.format, + None, native, + crate::git_cache::CacheOwnership { + work: owner.clone(), + cleanup: Some(owner.clone()), + }, ) .await?; - cache.download_native(store, descriptor).await?; + cache + .download_native_owned(store, descriptor, owner) + .await?; let pinned = Arc::clone(&cache); let input_claim = cache .native @@ -166,9 +238,15 @@ impl PhysicalVerifier { }) .await??; validate_native(Arc::clone(&cache), descriptor, limits.native_timeout).await?; - let native = CanonicalVerifier::new(&cache.git_dir(), descriptor.format, &cache.native)?; + let native = CanonicalVerifier::new_owned( + &cache.git_dir(), + descriptor.format, + &cache.native, + cache.clone(), + )?; Ok(Self { store: store.clone(), + context: None, native: Some(native), binding: Arc::new(binding), cache, @@ -190,6 +268,7 @@ impl PhysicalVerifier { &mut self, object_count: u32, ) -> Result, PhysicalError> { + self.ensure_live()?; if self.failed { return Err(PhysicalError::Integrity); } @@ -203,7 +282,9 @@ impl PhysicalVerifier { let root = self.root.clone(); let budget = self.budget.clone(); let limits = self.limits.metadata; + let cache = Arc::clone(&self.cache); let mut builder = tokio::task::spawn_blocking(move || { + let _pin = cache; MetadataBuilder::new(&root, budget, identity, limits) }) .await??; @@ -225,8 +306,10 @@ impl PhysicalVerifier { return Err(PhysicalError::Integrity); } let mut witnesses = Vec::with_capacity(ids.len()); - let edges = spool::EdgeSpool::new(&self.root, self.budget.clone()); + let edges = + spool::EdgeSpool::new_owned(&self.root, self.budget.clone(), self.cache.clone()); for oid in ids { + self.ensure_live()?; let witness = self .native .as_mut() @@ -257,11 +340,13 @@ impl PhysicalVerifier { self.chain = fold_shard(self.chain, self.shards, segment.descriptor()); self.shards = self.shards.checked_add(1).ok_or(PhysicalError::Integrity)?; self.next_ordinal = end; + self.ensure_live()?; self.failed = false; Ok(segment) } pub async fn finish(mut self) -> Result { + self.ensure_live()?; if self.failed || self.next_ordinal != self.descriptor.object_count || self.shards == 0 { return Err(PhysicalError::Integrity); } @@ -269,6 +354,7 @@ impl PhysicalVerifier { tokio::time::timeout(self.limits.native_timeout, native.finish()) .await .map_err(|_| GitHttpError::Timeout)??; + self.ensure_live()?; Ok(PhysicalPackWitness { store: self.store, native: self.descriptor, @@ -276,6 +362,12 @@ impl PhysicalVerifier { metadata_digest: self.chain, }) } + fn ensure_live(&self) -> Result<(), PhysicalError> { + if let Some(context) = &self.context { + context.ensure_live()?; + } + Ok(()) + } } fn pack_path(cache: &GitCache, descriptor: NativePackDescriptor) -> PathBuf { diff --git a/crates/canopy-server/src/packs/verification/spool.rs b/crates/canopy-server/src/packs/verification/spool.rs index 6b2f445a..111a05fb 100644 --- a/crates/canopy-server/src/packs/verification/spool.rs +++ b/crates/canopy-server/src/packs/verification/spool.rs @@ -104,6 +104,7 @@ struct Storage { file: Option, bytes: u64, failed: bool, + _owner: crate::git_objects::ReadOwner, } /// One append-only dependency file for a bounded native-object batch. Each @@ -117,6 +118,13 @@ pub(super) struct EdgeSpool { } impl EdgeSpool { pub(super) fn new(root: &Path, budget: DiskBudget) -> Self { + Self::new_owned(root, budget, Arc::new(())) + } + pub(super) fn new_owned( + root: &Path, + budget: DiskBudget, + owner: crate::git_objects::ReadOwner, + ) -> Self { Self { root: root.to_owned(), budget, @@ -124,6 +132,7 @@ impl EdgeSpool { file: None, bytes: 0, failed: false, + _owner: owner, })), writing: Arc::new(AtomicBool::new(false)), } diff --git a/crates/canopy-server/src/repository_http/browse.rs b/crates/canopy-server/src/repository_http/browse.rs index de0ac3c8..5f22810c 100644 --- a/crates/canopy-server/src/repository_http/browse.rs +++ b/crates/canopy-server/src/repository_http/browse.rs @@ -2,6 +2,10 @@ use super::*; use crate::git_read::{ReadError, Reader}; use std::time::Duration; +// A legal 65,535-byte ref cursor can expand sixfold in JSON escapes. Keep +// requests bounded while allowing clients to continue every supported ref page. +const REQUEST_BYTES: usize = 512 * 1024; + #[derive(Deserialize)] #[serde(deny_unknown_fields)] struct Input { @@ -75,7 +79,7 @@ async fn serve( }; let body = match tokio::time::timeout( Duration::from_secs(30), - to_bytes(request.into_body(), 32 * 1024), + to_bytes(request.into_body(), REQUEST_BYTES), ) .await { diff --git a/crates/canopy-server/src/server/catalog_initialization.rs b/crates/canopy-server/src/server/catalog_initialization.rs new file mode 100644 index 00000000..d4cc3b5f --- /dev/null +++ b/crates/canopy-server/src/server/catalog_initialization.rs @@ -0,0 +1,546 @@ +//! Certified empty-catalog initialization before exposing a repository route. +use crate::{ + ObjectFormat, RepositoryCell, RepositoryModule, + packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}, + metadata::MetadataLimits, + publication::{ + BeginRequest, CatalogPreparation, CheckInitializedCatalog, CheckPreparation, + CustodyAction, CustodyError, DEFAULT_LEASE_MS, GenerationFact, InitializationReply, + LeaseCheck, LeaseRequest, MaintenanceRequest, PreparationAuthority, + PreparationBaseResolver, PreparationDenial, PreparationReply, PreparationToken, + PreparedCustody, PublicationError, RegisteredCustody, RegisteredRootRecovery, + TerminalReleaseReply, + }, + }, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use cellule_runtime::{ + CellClient, Error, InvocationError, Receipt, + identity::IncarnationId, + primitives::sql::{SqlBatch, SqlCell, SqlStatement, SqlValue}, + registry::OwnerFence, +}; +use object_store::ObjectStore; +use std::{path::Path, sync::Arc}; + +type Failure = Box; + +fn request(repository: &RepositoryCell, owner: &str) -> BeginRequest { + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.repository.initialization.v1\0"); + for field in [ + repository.target.tenant().as_bytes().as_slice(), + repository.target.application().as_bytes().as_slice(), + repository.id.as_slice(), + repository.object_format.as_str().as_bytes(), + owner.as_bytes(), + ] { + hash.update(&(field.len() as u64).to_be_bytes()); + hash.update(field); + } + let request_digest = *hash.finalize().as_bytes(); + let mut operation = [0; 16]; + operation.copy_from_slice(&request_digest[..16]); + operation[0] |= 1; + BeginRequest { + repository: repository.id, + operation, + request_digest, + actor: owner.into(), + lease_ms: DEFAULT_LEASE_MS, + } +} + +fn initialization_custody(action: &CustodyAction, input: &BeginRequest) -> bool { + match action { + CustodyAction::BeginPreparation(request) => request == input, + CustodyAction::ClaimPreparation(request) | CustodyAction::RenewPreparation(request) => { + let check = &request.check; + check.actor == input.actor + && check.token.repository == input.repository + && check.token.operation == input.operation + && check.token.request_digest == input.request_digest + } + _ => false, + } +} +async fn custody_command( + client: &CellClient, + target: &cellule_runtime::CellTarget, + action: CustodyAction, +) -> Result, Failure> { + let prepared = + PreparedCustody::prepare(client, target, action, super::mutation_identity()?).await?; + let registered = prepared + .register(client, super::mutation_identity()?) + .await?; + Ok(registered.recover_preparation(client).await?) +} + +async fn verify( + fact: GenerationFact, + store: &ArtifactStore, + format: ObjectFormat, +) -> Result<(), Failure> { + crate::packs::publication::verify_initial_catalog(fact, store, format).await?; + Ok(()) +} + +/// The caller's tracked cold-transition task owns this work through cancellation. +/// Ready repositories only observe their immutable initialization; they cannot +/// reconstruct missing ownership or publish a new empty catalog during restore. +pub(super) struct InitializationCustody { + pub authority: PreparationAuthority, + pub maintenance: MaintenanceRequest, +} + +pub(super) async fn ensure( + custody: InitializationCustody, + repository: &RepositoryCell, + client: CellClient, + provider: Arc, + workspace: &Path, + budget: DiskBudget, + pending: bool, +) -> Result<(), Failure> { + let InitializationCustody { + authority, + maintenance, + } = custody; + let owner = maintenance.actor.as_str(); + let input = request(repository, owner); + let target = &repository.target; + let store = Arc::new(ArtifactStore::new(provider, repository.id)); + if let Some(fact) = client + .query::(target, None, input.clone()) + .await? + .output + { + return verify_and_retire(repository, &client, &store, &input, fact, &maintenance).await; + } + if !pending { + return Err(Error::Command("ready repository has no certified initialization").into()); + } + let started_at = std::time::Instant::now(); + let recovered = + RegisteredRootRecovery::load_initialization(&client, target, &store, &input).await?; + let claim = if let Some(ref recovered) = recovered { + match recovered + .recover_initialization(&client, &store, &authority) + .await + { + Ok(committed) => { + let InitializationReply::Initialized(fact) = committed.output else { + return Err(Error::Command("invalid recovered initialization reply").into()); + }; + verify(*fact, &store, repository.object_format).await?; + retire(recovered, client, &store, &maintenance).await?; + return Ok(()); + } + Err(PublicationError::Initialization(InvocationError::Rejected(ref value))) + if matches!( + value.output, + InitializationReply::Denied( + PreparationDenial::Stale | PreparationDenial::Expired + ) + ) => + { + Some(LeaseCheck { + token: recovered.token(), + actor: owner.into(), + }) + } + Err(error) => return Err(error.into()), + } + } else { + None + }; + // Discover the exact latest custody phase before constructing another SDK + // identity. Both accepted and denied Begin/Claim/Renew survive process loss. + let custody = startup_head(&client, target, &input, &authority).await?; + let refused_attempt = claim.as_ref().map(|check| check.token); + let action = if let Some(check) = claim { + CustodyAction::ClaimPreparation(LeaseRequest { + check, + lease_ms: DEFAULT_LEASE_MS, + }) + } else { + CustodyAction::BeginPreparation(input.clone()) + }; + if let Some(ref custody) = custody + && !initialization_custody(&custody.action()?, &input) + { + return Err(Error::Command("initialization custody context differs").into()); + } + // A stop closes registration, not execution: never manufacture a receipt or + // denial for the original. Observe the current operation after the separate + // authenticated stop receipt, then let the new receiver authorize a successor. + let stopped = custody.as_ref().and_then(RegisteredCustody::stop_fact); + let action = if let Some(stopped) = stopped { + match observed_attempt(&client, repository, &input, stopped.receipt).await? { + Some(check) => CustodyAction::ClaimPreparation(LeaseRequest { + check, + lease_ms: DEFAULT_LEASE_MS, + }), + None => CustodyAction::BeginPreparation(input.clone()), + } + } else { + action + }; + let replay = custody.as_ref().filter(|saved| saved.stop_fact().is_none()); + let result = if let Some(custody) = replay { + custody + .recover_preparation(&client) + .await + .map_err(|error| Box::new(error) as Failure) + } else { + custody_command(&client, target, action.clone()).await + }; + let started = match result { + Ok(started) => started, + Err(error) => { + let error = match error.downcast::>() { + Ok(error) => *error, + Err(error) => return Err(error), + }; + if let InvocationError::Rejected(ref rejected) = error + && matches!( + rejected.output, + PreparationReply::Denied(PreparationDenial::Stale | PreparationDenial::Expired) + ) + { + let prior = replay + .map(RegisteredCustody::action) + .transpose()? + .unwrap_or(action); + let check = match prior { + CustodyAction::ClaimPreparation(request) + | CustodyAction::RenewPreparation(request) => request.check, + CustodyAction::BeginPreparation(_) => { + prior_attempt(&client, repository, &input, rejected.receipt).await? + } + _ => { + return Err(Error::Command("initialization custody purpose differs").into()); + } + }; + custody_command( + &client, + target, + CustodyAction::ClaimPreparation(LeaseRequest { + check, + lease_ms: DEFAULT_LEASE_MS, + }), + ) + .await? + } else { + // Only a known conflict can observe a winning initialization. + // Unknown/expired SDK evidence never authorizes a new Begin. + if matches!(&error, InvocationError::Rejected(value) if value.output == PreparationReply::Denied(PreparationDenial::Conflict)) + && let Some(fact) = client + .query::(target, None, input.clone()) + .await? + .output + { + return verify_and_retire( + repository, + &client, + &store, + &input, + fact, + &maintenance, + ) + .await; + } + return Err(error.into()); + } + } + }; + let PreparationReply::Granted(ref original) = started.output else { + return Err(Error::Command("initialization custody grant absent").into()); + }; + // A historical result is knowledge only. Fresh custody and the actual owner + // are required before using its token; never restart its recorded clock. + let check = LeaseCheck { + token: original.token, + actor: owner.into(), + }; + let current = + if original.token.owner == maintenance.owner && refused_attempt != Some(original.token) { + client + .query::(target, Some(started.receipt), check.clone()) + .await? + .output + } else { + None + }; + let started = if let Some(current) = current { + if current.token != original.token + || current.base != original.base + || current.format != original.format + { + return Err(Error::Command("initialization custody result differs").into()); + } + started + } else { + custody_command( + &client, + target, + CustodyAction::ClaimPreparation(LeaseRequest { + check, + lease_ms: DEFAULT_LEASE_MS, + }), + ) + .await? + }; + if let Some(recovered) = recovered { + retire(&recovered, client.clone(), &store, &maintenance).await?; + } + let PreparationReply::Granted(lease) = started.output else { + return Err(Error::Command("repository initialization admission denied").into()); + }; + let check = LeaseCheck { + token: lease.token, + actor: owner.into(), + }; + let indexes = Arc::new(CatalogIndexes::new( + Arc::clone(&store), + repository.object_format, + )); + let files = Arc::new(CatalogFiles::new( + workspace, + budget.clone(), + Arc::clone(&store), + repository.object_format, + CatalogFileLimits::default(), + )?); + let base = Arc::new( + PreparationBaseResolver::open( + client.clone(), + target.clone(), + check, + indexes, + files, + Some(started.receipt), + authority.clone(), + ) + .await?, + ); + let prepared = Arc::new( + CatalogPreparation::new(workspace, budget, base, MetadataLimits::default()) + .await? + .finish() + .await?, + ); + let ready = prepared + .ready_initialization(super::mutation_identity()?) + .await?; + let registered = ready + .persist_recovery(&store, super::mutation_identity()?) + .await?; + let committed = ready.complete(®istered, &store).await?; + let InitializationReply::Initialized(fact) = committed.output else { + return Err(Error::Command("repository initialization publication denied").into()); + }; + verify(*fact, &store, repository.object_format).await?; + retire(®istered, client, &store, &maintenance).await?; + tracing::debug!(repository = %hex::encode(repository.id), elapsed_seconds = started_at.elapsed().as_secs_f64(), "certified repository catalog initialized"); + Ok(()) +} + +/// Already owned by the account-bounded, tracked cold transition. No background +/// outbox or new native work is introduced; ambiguous original outcomes retain +/// their exact identity, and only receiver-accepted closure permits a successor. +async fn startup_head( + client: &CellClient, + target: &cellule_runtime::CellTarget, + input: &BeginRequest, + authority: &PreparationAuthority, +) -> Result, Failure> { + let Some(saved) = RegisteredCustody::load_latest(client, target, input.operation).await? else { + return Ok(None); + }; + if !initialization_custody(&saved.action()?, input) { + return Err(Error::Command("initialization custody context differs").into()); + } + if saved.closed() || saved.evidence().identity().expires_at_ms >= super::unix_now_ms()? { + return Ok(Some(saved)); + } + // Journal knowledge precedes SDK expiry. Never retire an already known + // grant/denial merely because this previously loaded DTO has no phase. + if !matches!(saved.recover_preparation(client).await, + Err(InvocationError::Pending(ref evidence)) if **evidence == *saved.evidence()) + { + return Ok(Some(saved)); + } + match saved + .ready_stop(client.clone(), super::mutation_identity()?, authority) + .await + { + Ok(ready) => { + let outcome = ready.complete_tracked().await?; + if outcome.original != *saved.evidence() { + return Err(Error::Command("initialization retirement original differs").into()); + } + } + Err(CustodyError::Stopped(_)) => {} // Another helper already recorded closure. + Err(error) => return Err(error.into()), + } + let current = RegisteredCustody::load_latest(client, target, input.operation) + .await? + .ok_or(Error::Command("initialization custody disappeared"))?; + if !initialization_custody(¤t.action()?, input) + || (current.evidence() == saved.evidence() && !current.closed()) + { + return Err(Error::Command("initialization retirement not established").into()); + } + Ok(Some(current)) +} + +async fn verify_and_retire( + repository: &RepositoryCell, + client: &CellClient, + store: &ArtifactStore, + input: &BeginRequest, + fact: GenerationFact, + maintenance: &MaintenanceRequest, +) -> Result<(), Failure> { + verify(fact, store, repository.object_format).await?; + if let Some(recovered) = + RegisteredRootRecovery::load_initialization(client, &repository.target, store, input) + .await? + { + retire(&recovered, client.clone(), store, maintenance).await?; + } + Ok(()) +} + +async fn retire( + registered: &RegisteredRootRecovery, + client: CellClient, + store: &ArtifactStore, + maintenance: &MaintenanceRequest, +) -> Result<(), Failure> { + let result = registered + .ready_terminal_release( + client, + store, + maintenance.clone(), + super::mutation_identity()?, + ) + .await? + .complete() + .await?; + if result.output != TerminalReleaseReply::Released { + return Err(Error::Command("initialization retirement denied").into()); + } + Ok(()) +} + +fn fixed(value: &SqlValue) -> Result<[u8; N], Error> { + match value { + SqlValue::Blob(bytes) => bytes + .as_slice() + .try_into() + .map_err(|_| Error::Command("invalid prior initialization binding")), + _ => Err(Error::Command("invalid prior initialization binding")), + } +} +async fn prior_attempt( + client: &CellClient, + repository: &RepositoryCell, + input: &BeginRequest, + minimum: Receipt, +) -> Result { + observed_attempt(client, repository, input, minimum) + .await? + .ok_or_else(|| Error::Command("prior initialization attempt absent").into()) +} +async fn observed_attempt( + client: &CellClient, + repository: &RepositoryCell, + input: &BeginRequest, + minimum: Receipt, +) -> Result, Failure> { + let sql = SqlCell::::new(client.clone(), repository.target.clone())?; + let observed = sql.query(Some(minimum), SqlBatch { statements: vec![SqlStatement { + sql: "SELECT r.repository_id,r.object_format,r.owner,o.actor,o.request_digest,o.generation,o.incarnation,o.owner_epoch,o.admission_sequence,o.artifact_operation FROM repository_identity r LEFT JOIN catalog_operations o ON o.id=?1 WHERE r.singleton=1".into(), + parameters: vec![SqlValue::Blob(input.operation.to_vec())], + }] }).await?; + let Some( + [ + SqlValue::Blob(id), + SqlValue::Text(format), + SqlValue::Text(owner), + actor, + digest, + generation, + incarnation, + epoch, + sequence, + operation, + ], + ) = observed + .output + .first() + .and_then(|set| set.rows.first()) + .map(Vec::as_slice) + else { + return Err(Error::Command("initialization identity absent or malformed").into()); + }; + if id.as_slice() != repository.id + || format != repository.object_format.as_str() + || owner != &input.actor + { + return Err(Error::Command("initialization identity differs").into()); + } + if [ + actor, + digest, + generation, + incarnation, + epoch, + sequence, + operation, + ] + .iter() + .all(|value| matches!(value, SqlValue::Null)) + { + return Ok(None); + } + let ( + SqlValue::Text(actor), + SqlValue::Blob(digest), + SqlValue::Integer(0), + SqlValue::Integer(sequence), + ) = (actor, digest, generation, sequence) + else { + return Err(Error::Command("prior initialization binding malformed").into()); + }; + if actor != &input.actor || digest.as_slice() != input.request_digest { + return Err(Error::Command("prior initialization binding differs").into()); + } + let attempt = u64::try_from(*sequence) + .map_err(|_| Error::Command("invalid prior initialization sequence"))?; + if attempt == 0 { + return Err(Error::Command("invalid prior initialization sequence").into()); + } + Ok(Some(LeaseCheck { + actor: input.actor.clone(), + token: PreparationToken { + repository: repository.id, + operation: input.operation, + request_digest: input.request_digest, + owner: OwnerFence { + incarnation: IncarnationId::from_bytes(fixed(incarnation)?), + epoch: u64::from_be_bytes(fixed(epoch)?), + }, + attempt, + artifact_operation: fixed(operation)?, + }, + })) +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/server/catalog_initialization/tests.rs b/crates/canopy-server/src/server/catalog_initialization/tests.rs new file mode 100644 index 00000000..519603ce --- /dev/null +++ b/crates/canopy-server/src/server/catalog_initialization/tests.rs @@ -0,0 +1,530 @@ +use super::*; +use crate::packs::publication::{ + PublicationBudget, PublicationCoordinator, PublicationLimits, PublicationOutcome, + PublicationState, +}; +use crate::{CanopyApplication, build_descriptor, repository_target}; +use cellule_app::{ApplicationHandle, CellApplication, CompiledApplication}; +use cellule_ltx::{CellReplica, Limits}; +use cellule_runtime::{ + ApplicationId, CellModule, CellRuntime, Resolution, SessionId, TenantId, + cell::{ + actor::CellHandle, + catalog::{CatalogEntry, CatalogRole, CellCatalog}, + worker::SqlWorkerPool, + }, + control::{Owner, authority::CellAuthority}, + ltx::CellStorageLayout, +}; +use cellule_store::Store; +use object_store::{memory::InMemory, path::Path as StorePath}; +use tokio::time::{Duration, timeout}; + +type TestResult = Result; +async fn expire(original: &RegisteredCustody) -> TestResult { + loop { + let expiry = original.evidence().identity().expires_at_ms; + let now = super::super::unix_now_ms()?; + if now > expiry { + return Ok(()); + } + tokio::time::sleep(Duration::from_millis(u64::try_from(expiry - now + 1)?)).await; + } +} +struct Fixture { + files: tempfile::TempDir, + repository: RepositoryCell, + client: CellClient, + runtime: CellRuntime, + handle: CellHandle, + authority: PreparationAuthority, + provider: Arc, + layout: CellStorageLayout, + replica: CellReplica, + application: Arc, + publication_budget: PublicationBudget, +} +impl Fixture { + async fn new(format: ObjectFormat) -> TestResult { + let application = Arc::new(CanopyApplication::compile(build_descriptor( + include_bytes!("../../../Cargo.toml"), + "startup-custody-test", + ))?); + let tenant = TenantId::from_bytes([91; 16]); + let application_id = ApplicationId::from_bytes([92; 16]); + let id = *uuid::Uuid::new_v4().as_bytes(); + let target = repository_target(tenant, application_id, id)?; + let provider: Arc = Arc::new(InMemory::new()); + let layout = CellStorageLayout::new( + Store::new(Arc::clone(&provider)), + StorePath::from("startup-custody"), + *application_id.as_bytes(), + ); + let registry = application.registry(); + let proof = CellCatalog::new(layout.clone(), tenant) + .provision(CatalogEntry::new( + &target, + CatalogRole::Sql, + registry + .module_code(RepositoryModule::NAME) + .ok_or("module")?, + 1, + )?) + .await?; + let incarnation = IncarnationId::from_bytes([93; 16]); + let session = SessionId::from_bytes([94; 16]); + let control_authority = CellAuthority::new(layout.clone()); + let control = control_authority + .create_initial( + &proof, + incarnation, + Owner { + session, + endpoint: "https://startup-custody.invalid".into(), + }, + ) + .await?; + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + )?; + let files = tempfile::TempDir::new()?; + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let handle = runtime + .bootstrap( + proof, + replica.clone(), + control_authority, + control, + files.path().join("repository.sqlite"), + |tx| { + tx.execute_batch(crate::REPOSITORY_SCHEMA)?; + Ok(()) + }, + ) + .await?; + let client = CellClient::local(registry.clone(), handle.clone()); + let app = + ApplicationHandle::new(client.clone(), application.clone(), tenant, application_id)?; + let repository = RepositoryCell::new(&app, target.clone(), id, format)?; + repository + .ensure_owner(super::super::mutation_identity()?, "owner") + .await?; + Ok(Self { + files, + repository, + client, + runtime, + handle, + authority: PreparationAuthority::local(layout.clone(), target), + provider, + layout, + replica, + application, + publication_budget: PublicationBudget::new(PublicationLimits::default())?, + }) + } + fn input(&self) -> BeginRequest { + request(&self.repository, "owner") + } + async fn boot(&self) -> TestResult { + ensure( + InitializationCustody { + authority: self.authority.clone(), + maintenance: MaintenanceRequest { + repository: self.repository.id, + actor: "owner".into(), + owner: self.handle.owner_fence(), + }, + }, + &self.repository, + self.client.clone(), + Arc::clone(&self.provider), + self.files.path(), + DiskBudget::new(64 << 20), + true, + ) + .await + } + async fn register(&self, action: CustodyAction, short: bool) -> TestResult { + let mut identity = super::super::mutation_identity()?; + if short { + identity.expires_at_ms = identity.issued_at_ms + 1_000; + } + let ready = + PreparedCustody::prepare(&self.client, &self.repository.target, action, identity) + .await?; + Ok(ready + .register(&self.client, super::super::mutation_identity()?) + .await?) + } + async fn stop(&self, original: &RegisteredCustody) -> TestResult { + expire(original).await?; + let queue = PublicationCoordinator::new( + self.repository.target.clone(), + PublicationLimits::default(), + self.publication_budget.clone(), + )?; + let ready = original + .ready_stop( + self.client.clone(), + super::super::mutation_identity()?, + &self.authority, + ) + .await?; + let ticket = queue.submit(ready).await?; + let PublicationState::Finished(Ok(PublicationOutcome::CustodyStop(outcome))) = + timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("wrong stop outcome".into()); + }; + assert!(outcome.stop.is_some()); + assert!(queue.close_and_drain().await.is_empty()); + Ok(()) + } + async fn restore(&mut self) -> TestResult { + let old = self.handle.owner_fence(); + self.handle.drain().await?; + self.runtime.shutdown().await?; + std::fs::remove_file(self.files.path().join("repository.sqlite"))?; + for name in ["repository.sqlite-wal", "repository.sqlite-shm"] { + let path = self.files.path().join(name); + if path.exists() { + std::fs::remove_file(path)?; + } + } + let session = SessionId::from_bytes([95; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(self.layout.clone()); + let target = &self.repository.target; + let idle = authority.load(target.cell_id()).await?.ok_or("idle")?; + let proof = CellCatalog::new(self.layout.clone(), target.tenant()) + .lookup(target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + proof, + self.replica.clone(), + authority, + idle, + self.files.path().join("restored.sqlite"), + Owner { + session, + endpoint: "https://startup-restored.invalid".into(), + }, + ) + .await?; + assert!(handle.owner_fence().epoch > old.epoch); + self.client = CellClient::local(self.application.registry(), handle.clone()); + let app = ApplicationHandle::new( + self.client.clone(), + self.application.clone(), + target.tenant(), + target.application(), + )?; + self.repository = RepositoryCell::new( + &app, + target.clone(), + self.repository.id, + self.repository.object_format, + )?; + self.runtime = runtime; + self.handle = handle; + Ok(()) + } + async fn edit(&self, sql: &'static str) -> TestResult { + self.handle + .execute( + super::super::mutation_identity()?, + cellule_runtime::Digest::from_bytes(*blake3::hash(sql.as_bytes()).as_bytes()), + super::super::unix_now_ms()?, + sql.len(), + 0, + move |tx| { + tx.execute_batch(sql)?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + Ok(()) + } + async fn assert_original_preserved(&self, original: &RegisteredCustody) -> TestResult { + assert!( + matches!(original.recover_preparation(&self.client).await, Err(InvocationError::Pending(value)) if *value == *original.evidence()) + ); + assert!(matches!( + self.client.resolve(original.evidence()).await?, + Resolution::Expired + )); + let history = self.handle.query(0, 8, |c| Ok(c.query_row("SELECT count(*) FROM catalog_custody_commands WHERE phase IS NULL AND stopped IS NOT NULL", [], |r| r.get::<_, u64>(0))?.to_be_bytes().to_vec())).await?; + assert_eq!( + u64::from_be_bytes(history.try_into().map_err(|_| "count")?), + 1 + ); + Ok(()) + } +} + +#[tokio::test] +async fn stopped_startup_begin_can_initialize_without_inventing_an_original_result() -> TestResult { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let original = f + .register(CustodyAction::BeginPreparation(f.input()), true) + .await?; + f.stop(&original).await?; + f.boot().await?; + f.assert_original_preserved(&original).await?; + let head = + RegisteredCustody::load_latest(&f.client, &f.repository.target, f.input().operation) + .await? + .ok_or("head")?; + assert_ne!(head.evidence(), original.evidence()); + assert!(matches!(head.action()?, CustodyAction::BeginPreparation(_))); + assert!(head.settled()); + f.boot().await?; + assert_eq!( + RegisteredCustody::load_latest(&f.client, &f.repository.target, f.input().operation) + .await? + .ok_or("head")? + .evidence(), + head.evidence() + ); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn expired_unexecuted_startup_begin_is_retired_by_the_admitted_transition() -> TestResult { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let original = f + .register(CustodyAction::BeginPreparation(f.input()), true) + .await?; + expire(&original).await?; + f.boot().await?; + f.assert_original_preserved(&original).await?; + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn stopped_startup_claim_and_renew_use_current_attempt_after_real_owner_restore() -> TestResult +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for renew in [false, true] { + for stop_before_restore in [false, true] { + let mut f = Fixture::new(format).await?; + let begin = f + .register(CustodyAction::BeginPreparation(f.input()), false) + .await?; + let committed = begin.recover_preparation(&f.client).await?; + let PreparationReply::Granted(lease) = committed.output else { + return Err("begin grant".into()); + }; + let prior = lease.token; + let request = LeaseRequest { + check: LeaseCheck { + token: prior, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }; + let action = if renew { + CustodyAction::RenewPreparation(request) + } else { + CustodyAction::ClaimPreparation(request) + }; + let original = f.register(action, true).await?; + if stop_before_restore { + f.stop(&original).await?; + } else { + expire(&original).await?; + } + f.restore().await?; + f.boot().await?; + f.assert_original_preserved(&original).await?; + let head = RegisteredCustody::load_latest( + &f.client, + &f.repository.target, + f.input().operation, + ) + .await? + .ok_or("head")?; + let CustodyAction::ClaimPreparation(request) = head.action()? else { + return Err("successor must claim observed attempt".into()); + }; + assert_eq!(request.check.token, prior); + let PreparationReply::Granted(next) = + head.recover_preparation(&f.client).await?.output + else { + return Err("successor grant".into()); + }; + assert_eq!(next.token.owner, f.handle.owner_fence()); + assert_ne!(next.token, prior); + f.runtime.shutdown().await?; + } + } + } + Ok(()) +} + +#[tokio::test] +async fn known_startup_result_precedes_sdk_expiry_and_unexpired_absence_reuses_original() +-> TestResult { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for execute in [false, true] { + let f = Fixture::new(format).await?; + let original = f + .register(CustodyAction::BeginPreparation(f.input()), execute) + .await?; + let prior = if execute { + Some(original.recover_preparation(&f.client).await?) + } else { + None + }; + if execute { + expire(&original).await?; + } + f.boot().await?; + let head = RegisteredCustody::load_latest( + &f.client, + &f.repository.target, + f.input().operation, + ) + .await? + .ok_or("head")?; + assert_eq!(head.evidence(), original.evidence()); + assert!(head.stop_fact().is_none()); + let result = head.recover_preparation(&f.client).await?; + if let Some(prior) = prior { + assert_eq!(result.receipt, prior.receipt); + } + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn stopped_startup_cannot_turn_inconsistent_binding_into_absence() -> TestResult { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for sql in [ + "UPDATE catalog_operations SET actor='other'", + "UPDATE catalog_operations SET request_digest=zeroblob(32)", + "UPDATE repository_identity SET owner='other'", + ] { + let f = Fixture::new(format).await?; + let begin = f + .register(CustodyAction::BeginPreparation(f.input()), false) + .await?; + let PreparationReply::Granted(lease) = + begin.recover_preparation(&f.client).await?.output + else { + return Err("begin grant".into()); + }; + let original = f + .register( + CustodyAction::RenewPreparation(LeaseRequest { + check: LeaseCheck { + token: lease.token, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }), + true, + ) + .await?; + f.stop(&original).await?; + f.edit(sql).await?; + assert!(f.boot().await.is_err()); + let latest = RegisteredCustody::load_latest( + &f.client, + &f.repository.target, + f.input().operation, + ) + .await? + .ok_or("head")?; + assert_eq!(latest.evidence(), original.evidence()); + assert!(!latest.settled()); + f.assert_original_preserved(&original).await?; + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn unavailable_owner_keeps_expired_startup_original_unretired_until_authority_returns() +-> TestResult { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for corrupt in [false, true] { + let f = Fixture::new(format).await?; + let original = f + .register(CustodyAction::BeginPreparation(f.input()), true) + .await?; + expire(&original).await?; + let path = f + .layout + .control_path(f.repository.target.cell_id().as_bytes()); + let (control, _) = f.layout.store().get_with_etag(&path).await?; + if corrupt { + f.layout + .store() + .put_overwrite(&path, bytes::Bytes::from_static(b"invalid control")) + .await?; + } else { + f.layout.store().delete(&path).await?; + } + let error = f + .boot() + .await + .expect_err("missing owner must refuse retirement"); + assert!(matches!( + error.downcast_ref::(), + Some(CustodyError::Owner(_)) + )); + let head = RegisteredCustody::load_latest( + &f.client, + &f.repository.target, + f.input().operation, + ) + .await? + .ok_or("head")?; + assert_eq!(head.evidence(), original.evidence()); + assert!(!head.closed()); + f.layout.store().put_overwrite(&path, control).await?; + f.boot().await?; + f.assert_original_preserved(&original).await?; + f.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn startup_does_not_retire_another_purpose_at_its_logical_operation() -> TestResult { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let original = f + .register(CustodyAction::BeginStaging(f.input()), true) + .await?; + expire(&original).await?; + assert!(f.boot().await.is_err()); + let head = + RegisteredCustody::load_latest(&f.client, &f.repository.target, f.input().operation) + .await? + .ok_or("head")?; + assert_eq!(head.evidence(), original.evidence()); + assert!(!head.closed()); + f.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/server/lifecycle.rs b/crates/canopy-server/src/server/lifecycle.rs index 1b81570a..2476c597 100644 --- a/crates/canopy-server/src/server/lifecycle.rs +++ b/crates/canopy-server/src/server/lifecycle.rs @@ -111,9 +111,10 @@ impl CanopyServer { } impl RunningServer { - async fn shutdown(mut self) -> Result<(), ServerError> { + pub(super) async fn shutdown(mut self) -> Result<(), ServerError> { self.native.close(); self.maintenance_stop.cancel(); + self.repositories.recovery_scans.close(); self.ingress_stop.cancel(); let serving = self.serving.await; let ssh_serving = if let Some(task) = self.ssh_serving { @@ -122,8 +123,10 @@ impl RunningServer { None }; self.listeners.stop_ingress(); + self.repositories.drain_serving().await; self.tasks.close(); self.tasks.wait().await; + self.repositories.drain_recovery().await; // Detached native reapers and blocking verifiers outlive their request // observers. Keep Cell authority, heartbeat and workspace until every // admitted owner releases its claim. Uncertain drain stays pending. diff --git a/crates/canopy-server/src/server/mod.rs b/crates/canopy-server/src/server/mod.rs index a91c9c40..3ab1e1e9 100644 --- a/crates/canopy-server/src/server/mod.rs +++ b/crates/canopy-server/src/server/mod.rs @@ -47,6 +47,7 @@ use crate::{ }; mod catalog_admission; +mod catalog_initialization; mod discovery; mod lifecycle; mod listeners; @@ -67,6 +68,10 @@ pub(crate) const RENEW_INTERVAL: Duration = Duration::from_secs(3); #[derive(Debug, thiserror::Error)] pub enum ServerError { + #[error("packed repository recovery failed")] + CatalogRecovery(#[source] Box), + #[error("packed repository initialization failed")] + CatalogInitialization(#[source] Box), #[error("invalid SSH host key")] SshKey(#[source] Box), #[error("Cellule runtime failed")] @@ -172,6 +177,7 @@ struct RunningServer { listeners: listeners::ListenerReservations, local: Arc, native: crate::native_resources::NativeResources, + repositories: Arc, } pub(crate) struct RepositoryManager { @@ -197,7 +203,17 @@ pub(crate) struct RepositoryManager { residency_admission: AccountAdmission, transfers: AccountAdmission, tasks: TaskTracker, - maintenance_stop: CancellationToken, + publication_budget: crate::packs::publication::PublicationBudget, + recovery_scans: crate::packs::publication::RecoveryScanBudget, + serving_reads: crate::packs::publication::ServingReadBudget, + serving_stop: CancellationToken, + #[cfg(test)] + serving_construction_gate: Mutex< + Option<( + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + )>, + >, } pub(crate) enum MembershipOutcome { @@ -438,12 +454,13 @@ impl RunningServer { listeners.http = Some(reservation); socket }); + let store = Store::new(Arc::clone(&raw_store)); + crate::deployment::validate_service_root(&store, &config.store_prefix).await?; let data_dir = config.data_dir.clone(); let local = Arc::new( tokio::task::spawn_blocking(move || workspace::Workspace::open(&data_dir)).await??, ); listeners.workspace = Some(Arc::clone(&local)); - let store = Store::new(Arc::clone(&raw_store)); storage::probe(&store, &config.store_prefix.clone().join("canopy-probe")).await?; let application = Arc::new(CanopyApplication::compile(build_descriptor( include_bytes!("../../../../Cargo.lock"), @@ -638,7 +655,20 @@ impl RunningServer { "account repository activations", ), tasks: tasks.clone(), - maintenance_stop: maintenance_stop.clone(), + publication_budget: crate::packs::publication::PublicationBudget::new( + crate::packs::publication::PublicationLimits::default(), + ) + .map_err(|error| ServerError::CatalogRecovery(Box::new(error)))?, + recovery_scans: crate::packs::publication::RecoveryScanBudget::new( + 8, + tasks.clone(), + ) + .map_err(|error| ServerError::CatalogRecovery(Box::new(error)))?, + serving_reads: crate::packs::publication::ServingReadBudget::new(64, tasks.clone()) + .map_err(|error| ServerError::CatalogRecovery(Box::new(error)))?, + serving_stop: CancellationToken::new(), + #[cfg(test)] + serving_construction_gate: Mutex::new(None), }); let api = Arc::new(RepositoryHttp::new(Arc::clone(&manager), tasks.clone())); deployment.require_ready().await?; @@ -673,6 +703,7 @@ impl RunningServer { let ingress_stop = CancellationToken::new(); let ssh_serving = match (ssh_config, ssh_listener) { (Some(config), Some(listener)) => { + let manager = Arc::clone(&manager); let stop = ingress_stop.clone(); let tasks = tasks.clone(); let release = release_stop.clone(); @@ -713,6 +744,7 @@ impl RunningServer { listeners, local, native, + repositories: manager, }) } } diff --git a/crates/canopy-server/src/server/peer.rs b/crates/canopy-server/src/server/peer.rs index 72bda0d8..abd7774a 100644 --- a/crates/canopy-server/src/server/peer.rs +++ b/crates/canopy-server/src/server/peer.rs @@ -118,10 +118,31 @@ impl NodePeer { .is_some_and(|owner| owner.session() != self.0.session)) } + pub(crate) async fn current_owner_fence( + &self, + target: &CellTarget, + ) -> Result { + self.live_binding(target) + .await? + .map(|(_, fence)| fence) + .ok_or(Error::Fenced.into()) + } + async fn live_owner( &self, target: &CellTarget, ) -> Result, ServerError> { + Ok(self + .live_binding(target) + .await? + .map(|(advertisement, _)| advertisement)) + } + + async fn live_binding( + &self, + target: &CellTarget, + ) -> Result, ServerError> + { let authority = CellAuthority::new(self.0.layout.clone()); let Some(control) = authority.load(target.cell_id()).await? else { return Ok(None); @@ -136,7 +157,10 @@ impl NodePeer { if live.advertisement().endpoint() != owner.endpoint { return Err(Error::Fenced.into()); } - Ok(Some(live.advertisement().clone())) + Ok(Some(( + live.advertisement().clone(), + control.value().owner_fence(), + ))) } pub(super) async fn ensure_directory(&self) -> Result<(), ServerError> { diff --git a/crates/canopy-server/src/server/residency/mod.rs b/crates/canopy-server/src/server/residency/mod.rs index 7f57850b..224cca71 100644 --- a/crates/canopy-server/src/server/residency/mod.rs +++ b/crates/canopy-server/src/server/residency/mod.rs @@ -28,9 +28,12 @@ use crate::{ http::GitHttpApi, repository_target, }; +mod recovery; +use recovery::RecoveryServices; pub(super) struct LoadedRepository { repository: Arc, + client: CellClient, gateway: Arc, name: String, router: Router, @@ -40,6 +43,7 @@ pub(super) struct LoadedRepository { local: bool, state: ResidencyState, slot: Arc, + recovery: Option>, } enum EvictionAction { @@ -197,7 +201,9 @@ impl RepositoryManager { // Remote cache ownership is disposable. Reacquire idle/expired Cell // authority locally before binding a new route after owner loss. let removed = self.loaded.lock().await.remove(&entry.repository_id); - reclaimed = removed.map(|repository| repository.slot); + if let Some(repository) = removed { + reclaimed = Some(repository.slot); + } } let state = self .loaded @@ -206,6 +212,7 @@ impl RepositoryManager { .get(&entry.repository_id) .map(|repository| repository.state); match state { + Some(ResidencyState::Releasing) => return Err(Error::CellDraining.into()), Some(ResidencyState::Released) => { reclaimed = Some(self.cleanup_released(entry.repository_id).await?); } @@ -266,7 +273,7 @@ impl RepositoryManager { SqlCellSpec { target: &target, module: RepositoryModule::NAME, - schema: include_str!("../../schema.sql"), + schema: crate::REPOSITORY_SCHEMA, destination: directory.join("repository.sqlite"), }, self.session, @@ -302,8 +309,13 @@ impl RepositoryManager { .await .get(&entry.repository_id) .filter(|repository| !repository.initialized) - .map(|repository| Arc::clone(&repository.repository)); - if let Some(repository) = initialize { + .map(|repository| { + ( + Arc::clone(&repository.repository), + repository.client.clone(), + ) + }); + if let Some((repository, client)) = initialize { // Keep the acquired Cell through an uncertain initialization result. // A later request retries setup before any fast-path route is exposed. if entry.state == RepositoryState::Pending { @@ -321,6 +333,66 @@ impl RepositoryManager { "repository owner differs from directory", )); } + super::catalog_initialization::ensure( + super::catalog_initialization::InitializationCustody { + authority: crate::packs::publication::PreparationAuthority::node( + self.peer.clone(), + repository.target.clone(), + ), + maintenance: crate::packs::publication::MaintenanceRequest { + repository: entry.repository_id, + actor: entry.owner.clone(), + owner: self.peer.current_owner_fence(&repository.target).await?, + }, + }, + &repository, + client, + Arc::clone(&self.external_store), + self.local.path(), + self.disk_budget.clone(), + entry.state == RepositoryState::Pending, + ) + .await + .map_err(ServerError::CatalogInitialization)?; + } + let start = self + .loaded + .lock() + .await + .get(&entry.repository_id) + .filter(|repository| repository.local && repository.recovery.is_none()) + .map(|repository| { + ( + Arc::clone(&repository.repository), + repository.client.clone(), + ) + }); + if let Some((repository, client)) = start { + let recovery = + Arc::new(RecoveryServices::start(self, entry, &repository, client).await?); + #[cfg(test)] + if let Some((entered, proceed)) = self.serving_construction_gate.lock().await.take() { + let _ = entered.send(()); + let _ = proceed.await; + } + let rejected = { + let mut loaded = self.loaded.lock().await; + if self.serving_stop.is_cancelled() { + Some(ServerError::Runtime(Error::CellDraining)) + } else if let Some(resident) = loaded.get_mut(&entry.repository_id) { + resident.recovery = Some(recovery.clone()); + repository.attach_serving(&recovery.serving); + None + } else { + Some(ServerError::Repository("loaded repository is absent")) + } + }; + if let Some(error) = rejected { + // Construction is tracked by this residency owner. Join private + // pools, scanners and exact recovery before completing its task. + recovery.drain().await; + return Err(error); + } } let mut loaded = self.loaded.lock().await; let existing = loaded @@ -399,7 +471,7 @@ impl RepositoryManager { let Ok(transition) = self.transition_lock(id).await.try_lock_owned() else { continue; }; - if matches!(action, EvictionAction::Release { .. }) { + if !matches!(action, EvictionAction::Cleanup) { loaded .get_mut(&id) .ok_or(ServerError::Repository("eviction candidate is absent"))? @@ -438,7 +510,25 @@ impl RepositoryManager { } .into()); }; - let (cell, generation) = match action { + let recovery = self + .loaded + .lock() + .await + .get(&id) + .and_then(|repository| repository.recovery.as_ref().map(Arc::clone)); + if let Some(recovery) = recovery + && !recovery.quiesce().await + { + self.loaded + .lock() + .await + .get_mut(&id) + .ok_or(ServerError::Repository("eviction candidate is absent"))? + .state = ResidencyState::Serving; + rejected.insert(id); + continue; + } + let (cell, _) = match action { EvictionAction::DropRemote => { let removed = self.loaded.lock().await.remove(&id); return removed @@ -448,6 +538,35 @@ impl RepositoryManager { EvictionAction::Cleanup => return self.cleanup_released(id).await, EvictionAction::Release { cell, generation } => (cell, generation), }; + // The quiesced reads may have changed the runtime's idle generation + // since candidate selection. Reobserve actual settled authority; + // never substitute our earlier inventory or manufacture a handle. + let refreshed = match self.node.idle_transfer_candidates().await { + Ok(candidates) => candidates, + Err(error) => { + self.loaded + .lock() + .await + .get_mut(&id) + .ok_or(ServerError::Repository("eviction candidate is absent"))? + .state = ResidencyState::RefreshHandle; + return Err(error.into()); + } + }; + let generation = refreshed + .into_iter() + .find(|(candidate, _, _, _)| *candidate == cell) + .map(|(_, generation, _, _)| generation); + let Some(generation) = generation else { + self.loaded + .lock() + .await + .get_mut(&id) + .ok_or(ServerError::Repository("eviction candidate is absent"))? + .state = ResidencyState::RefreshHandle; + rejected.insert(id); + continue; + }; let mut result = self .node .release_idle_cell(cell, self.session, generation) @@ -516,7 +635,7 @@ impl RepositoryManager { slot: Arc, ) -> Result { let application = self.node.application_handle::( - client, + client.clone(), self.tenant, self.application, )?; @@ -538,29 +657,9 @@ impl RepositoryManager { ); let router = self.router_for(entry, Arc::clone(&gateway))?; let pin = Arc::new(()); - let weak_gateway = Arc::downgrade(&gateway); - let weak_pin = Arc::downgrade(&pin); - let stop = self.maintenance_stop.clone(); - self.tasks.spawn(async move { - let mut interval = tokio::time::interval(std::time::Duration::from_secs(60)); - interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - interval.tick().await; - loop { - tokio::select! { - () = stop.cancelled() => return, - _ = interval.tick() => {}, - } - let (Some(gateway), Some(_pin)) = (weak_gateway.upgrade(), weak_pin.upgrade()) else { return; }; - tokio::select! { - () = stop.cancelled() => return, - result = gateway.maintain() => { - if let Err(error) = result { tracing::warn!(error = ?error, "background Git maintenance failed; previous cache retained"); } - } - } - } - }); Ok(LoadedRepository { repository, + client, gateway, name: entry.name.clone(), router, @@ -570,6 +669,7 @@ impl RepositoryManager { local, state: ResidencyState::Serving, slot, + recovery: None, }) } diff --git a/crates/canopy-server/src/server/residency/recovery.rs b/crates/canopy-server/src/server/residency/recovery.rs new file mode 100644 index 00000000..bfba765a --- /dev/null +++ b/crates/canopy-server/src/server/residency/recovery.rs @@ -0,0 +1,219 @@ +//! Resident owners join discovery before release and retain exact uncertain work. +use super::*; +use crate::packs::catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}; +use crate::packs::publication::{ + CustodySupervisor, MaintenanceRequest, PreparationAuthority, PublicationCoordinator, + PublicationLimits, PublicationState, RecoveryScanLimits, RecoverySupervisor, ServingContext, + ServingPool, ServingPoolLimits, +}; +use canopy_object_storage::artifact::ArtifactStore; + +pub(super) struct RecoveryServices { + pub(super) coordinator: PublicationCoordinator, + pub(super) serving: Arc, + workers: Mutex>, +} +struct Workers { + roots: RecoverySupervisor, + custody: CustodySupervisor, +} +impl RecoveryServices { + pub(super) async fn start( + manager: &RepositoryManager, + entry: &RepositoryEntry, + repository: &RepositoryCell, + client: CellClient, + ) -> Result { + if manager.serving_stop.is_cancelled() { + return Err(Error::CellDraining.into()); + } + let target = repository.target.clone(); + let authority = PreparationAuthority::node(manager.peer.clone(), target.clone()); + let maintenance = MaintenanceRequest { + repository: entry.repository_id, + actor: entry.owner.clone(), + owner: manager.peer.current_owner_fence(&target).await?, + }; + let coordinator = PublicationCoordinator::new( + target.clone(), + PublicationLimits::default(), + manager.publication_budget.clone(), + ) + .map_err(|error| ServerError::CatalogRecovery(Box::new(error)))?; + let settings = manager + .recovery_scans + .settings(RecoveryScanLimits::default(), &entry.owner); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&manager.external_store), + entry.repository_id, + )); + let serving = Arc::new( + ServingPool::new( + ServingContext::new( + client.clone(), + target.clone(), + authority.clone(), + Arc::new(CatalogIndexes::new(store.clone(), entry.object_format)), + Arc::new( + CatalogFiles::new( + manager.local.path(), + manager.disk_budget.clone(), + store, + entry.object_format, + CatalogFileLimits::default(), + ) + .map_err(|error| ServerError::CatalogRecovery(Box::new(error)))? + .with_native( + manager + .native + .scope(crate::native_resources::NativeClass::Foreground), + ), + ), + manager.serving_reads.clone(), + entry.owner.clone(), + ) + .map_err(|error| ServerError::CatalogRecovery(Box::new(error)))?, + coordinator.clone(), + ServingPoolLimits::default(), + ) + .map_err(|error| ServerError::CatalogRecovery(Box::new(error)))?, + ); + let roots = match RecoverySupervisor::start_retiring( + client.clone(), + target.clone(), + ArtifactStore::new(Arc::clone(&manager.external_store), entry.repository_id), + coordinator.clone(), + settings.clone(), + authority.clone(), + maintenance, + ) { + Ok(roots) => roots, + Err(error) => { + serving.close_and_drain().await; + return Err(ServerError::CatalogRecovery(Box::new(error))); + } + }; + let custody = match CustodySupervisor::start( + client, + target, + coordinator.clone(), + settings, + authority, + ) { + Ok(custody) => custody, + Err(error) => { + // A partially constructed owner must join its first worker before + // giving up the residency transition or its workspace ownership. + let _ = roots.shutdown().await; + serving.close_and_drain().await; + return Err(ServerError::CatalogRecovery(Box::new(error))); + } + }; + Ok(Self { + coordinator, + serving, + workers: Mutex::new(Some(Workers { roots, custody })), + }) + } + + pub(super) async fn quiesce(&self) -> bool { + let mut workers = self.workers.lock().await; + if let Some(active) = workers.as_ref() { + tokio::join!(active.roots.pause(), active.custody.pause()); + } + let closed = match self.serving.quiesce().await { + Ok(closed) => closed, + Err(error) => { + tracing::warn!(?error, "serving drain refused repository eviction"); + false + } + }; + if !closed { + if let Some(active) = workers.as_ref() { + active.roots.resume(); + active.custody.resume(); + } + return false; + } + self.serving.close_and_drain().await; + join(workers.take()).await; + true + } + + pub(super) async fn drain(&self) { + self.serving.close_and_drain().await; + join(self.workers.lock().await.take()).await; + loop { + let pending = self.coordinator.close_and_drain().await; + if pending.is_empty() { + return; + } + for ticket in pending { + // Held final work belongs to its producer lifecycle. Do not + // activate/discard it or substitute an unknown outcome here. + if matches!(ticket.state(), PublicationState::Uncertain(_)) + && let Err(error) = ticket.recover().await + && error != crate::packs::publication::PublicationScheduleError::NotUncertain + { + tracing::warn!(?error, "exact repository recovery deferred during drain"); + } + } + tokio::time::sleep(std::time::Duration::from_secs(1)).await; + } + } +} +async fn join(workers: Option) { + if let Some(workers) = workers { + let (roots, custody) = tokio::join!(workers.roots.shutdown(), workers.custody.shutdown()); + // A failed discovery task is not evidence about admitted commands. + // Its control guard has drained; the coordinator remains owned below. + if let Err(error) = roots { + tracing::error!(?error, "repository root scanner failed"); + } + if let Err(error) = custody { + tracing::error!(?error, "repository custody scanner failed"); + } + } +} +impl RepositoryManager { + pub(in crate::server) async fn drain_serving(&self) { + let services: Vec<_> = { + let loaded = self.loaded.lock().await; + // Share the publication barrier with constructor registration. A + // late pool cannot escape this inventory or the node tracker join. + self.serving_stop.cancel(); + loaded + .values() + .filter_map(|repository| repository.recovery.as_ref().map(Arc::clone)) + .collect() + }; + for service in &services { + service.serving.close(); + } + futures_util::future::join_all( + services + .iter() + .map(|service| service.serving.close_and_drain()), + ) + .await; + self.serving_reads.close(); + } + pub(in crate::server) async fn drain_recovery(&self) { + self.drain_serving().await; + self.recovery_scans.close(); + self.publication_budget.close(); + // The existing residency cap bounds this inventory. No independent + // durable queue or historical-repository registry is introduced. + let services: Vec<_> = self + .loaded + .lock() + .await + .values() + .filter_map(|repository| repository.recovery.as_ref().map(Arc::clone)) + .collect(); + // One producer-held command must not prevent other repositories from + // resolving their exact originals. All futures remain owned by this + // drain; the existing residency cap bounds their concurrent inventory. + futures_util::future::join_all(services.iter().map(|service| service.drain())).await; + } +} diff --git a/crates/canopy-server/src/server/residency/tests.rs b/crates/canopy-server/src/server/residency/tests.rs index b6e57feb..c8125c54 100644 --- a/crates/canopy-server/src/server/residency/tests.rs +++ b/crates/canopy-server/src/server/residency/tests.rs @@ -1,6 +1,8 @@ use std::{collections::VecDeque, convert::Infallible, future::poll_fn}; use super::*; +mod recovery; +mod serving; struct Frames(VecDeque>); diff --git a/crates/canopy-server/src/server/residency/tests/recovery.rs b/crates/canopy-server/src/server/residency/tests/recovery.rs new file mode 100644 index 00000000..9c19146e --- /dev/null +++ b/crates/canopy-server/src/server/residency/tests/recovery.rs @@ -0,0 +1,384 @@ +use super::*; +use crate::packs::publication::{ + BeginRequest, CustodyAction, DEFAULT_LEASE_MS, LeaseCheck, PreparationAuthority, + PreparationReply, PreparationSession, PreparedCustody, PublicationError, PublicationOutcome, + PublicationState, RegisteredCustody, +}; +use crate::{ + ObjectFormat, + server::{RunningServer, ServerConfig, mutation_identity}, +}; +use cellule_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; +use cellule_runtime::{ApplicationId, SessionId, TenantId, identity::NodeId}; +use ed25519_dalek::SigningKey; +use object_store::{memory::InMemory, path::Path as StorePath}; +use tokio::time::{Duration, timeout}; + +type Result = std::result::Result>; + +pub(super) async fn server() -> Result<(RunningServer, tempfile::TempDir)> { + let files = tempfile::TempDir::new()?; + let server = RunningServer::start( + ServerConfig { + tenant: TenantId::from_bytes([71; 16]), + application: ApplicationId::from_bytes([72; 16]), + node: NodeId::from_bytes([73; 16]), + fleet: cellule_runtime::Digest::from_bytes([74; 32]), + image: cellule_runtime::Digest::from_bytes([75; 32]), + signing_key: SigningKey::from_bytes(&[76; 32]), + owner: "canopy".into(), + token: "local-recovery-test".into(), + public_url: "http://127.0.0.1".into(), + peer_endpoint: "https://recovery.test".into(), + peer_ca_pem: None, + listen: "127.0.0.1:0".parse()?, + ssh: None, + data_dir: files.path().join("node"), + store_prefix: StorePath::from("resident-recovery"), + local_disk_limit_bytes: 1 << 30, + native_limits: crate::native_resources::NativeLimits::default(), + max_active_repositories: 3, + }, + Arc::new(InMemory::new()), + None, + ) + .await?; + Ok((server, files)) +} +pub(super) async fn create( + manager: &Arc, + name: &str, + format: ObjectFormat, +) -> Result { + Ok(timeout(Duration::from_secs(10), async { + loop { + match manager.create(name, format).await { + Ok(entry) => return Ok(entry), + // Settlement/movement admission is transient. Retry the same + // directory reservation, never a different logical repository. + Err(ServerError::Runtime(Error::CellDraining | Error::Capacity(_))) => { + tokio::time::sleep(Duration::from_millis(20)).await; + } + Err(error) => return Err(error), + } + } + }) + .await??) +} +pub(super) async fn loaded( + manager: &RepositoryManager, + id: [u8; 16], +) -> Result<(Arc, CellClient, Arc)> { + let loaded = manager.loaded.lock().await; + let repository = loaded.get(&id).ok_or("repository not resident")?; + assert!(repository.local && repository.initialized); + Ok(( + Arc::clone(&repository.repository), + repository.client.clone(), + Arc::clone( + repository + .recovery + .as_ref() + .ok_or("production recovery absent")?, + ), + )) +} +fn request(id: [u8; 16]) -> BeginRequest { + BeginRequest { + repository: id, + operation: [81; 16], + request_digest: [82; 32], + actor: "canopy".into(), + lease_ms: DEFAULT_LEASE_MS, + } +} +async fn held_renewal( + manager: &RepositoryManager, + entry: &RepositoryEntry, +) -> Result { + let (repository, client, service) = loaded(manager, entry.repository_id).await?; + let prepared = PreparedCustody::prepare( + &client, + &repository.target, + CustodyAction::BeginPreparation(request(entry.repository_id)), + mutation_identity()?, + ) + .await?; + let saved = prepared.register(&client, mutation_identity()?).await?; + let committed = saved.recover_preparation(&client).await?; + let PreparationReply::Granted(lease) = committed.output else { + return Err("preparation refused".into()); + }; + let session = Arc::new( + PreparationSession::open( + client, + repository.target.clone(), + LeaseCheck { + token: lease.token, + actor: "canopy".into(), + }, + Some(committed.receipt), + PreparationAuthority::node(manager.peer.clone(), repository.target.clone()), + ) + .await?, + ); + Ok(service.coordinator.try_reserve( + session + .ready_renew(mutation_identity()?, DEFAULT_LEASE_MS) + .await?, + )?) +} + +#[tokio::test] +async fn production_scanner_retires_authentic_orphan_without_inventing_original_execution() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let manager = &server.repositories; + let entry = create(manager, "orphan", format).await?; + let (repository, client, _service) = loaded(manager, entry.repository_id).await?; + let mut identity = mutation_identity()?; + identity.expires_at_ms = identity.issued_at_ms + 1000; + let prepared = PreparedCustody::prepare( + &client, + &repository.target, + CustodyAction::BeginPreparation(request(entry.repository_id)), + identity, + ) + .await?; + let original = prepared.evidence().clone(); + let saved = prepared.register(&client, mutation_identity()?).await?; + drop((prepared, saved)); + let stopped = timeout(Duration::from_secs(10), async { + loop { + if let Some(saved) = + RegisteredCustody::load_latest(&client, &repository.target, [81; 16]).await? + && saved.stop_fact().is_some() + { + return Ok::<_, crate::packs::publication::CustodyError>(saved); + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await??; + assert_eq!(stopped.evidence(), &original); + assert!(!stopped.settled()); + let output = repository + .sql + .query( + None, + SqlBatch { + statements: vec![SqlStatement { + sql: "SELECT count(*) FROM catalog_operations WHERE id=?1".into(), + parameters: vec![SqlValue::Blob(vec![81; 16])], + }], + }, + ) + .await?; + assert_eq!(output.output[0].rows[0], vec![SqlValue::Integer(0)]); + drop((client, repository, stopped)); + server.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn production_eviction_joins_idle_scanners_and_preserves_busy_originals_and_restoration() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let manager = &server.repositories; + let one = create(manager, "one", format).await?; + let (_, _, service) = loaded(manager, one.repository_id).await?; + let held = held_renewal(manager, &one).await?; + assert_eq!(manager.publication_budget.stats().foreground, 1); + let two = create(manager, "two", format).await?; + let three = create(manager, "three", format).await?; + let four = create(manager, "four", format).await?; + assert_eq!(manager.loaded.lock().await.len(), 3); + assert!(manager.loaded.lock().await.contains_key(&one.repository_id)); + assert!(matches!(held.state(), PublicationState::Held)); + assert!(!service.coordinator.stats().await.closed); + assert_eq!(manager.publication_budget.stats().foreground, 1); + held.discard_held().await?; + let evicted = { + let loaded = manager.loaded.lock().await; + [two, three, four] + .into_iter() + .find(|entry| !loaded.contains_key(&entry.repository_id)) + .ok_or("no idle repository evicted")? + }; + // Production restores the same certified identity after terminal recovery + // retirement; no legacy SQL objects or compatibility decoder is added. + timeout(Duration::from_secs(10), async { + loop { + match manager + .load(ReadIdentity::Account("canopy"), evicted.clone()) + .await + { + Ok(route) => return Ok(route), + Err(ServerError::Runtime(Error::CellDraining | Error::Capacity(_))) => { + tokio::time::sleep(Duration::from_millis(20)).await + } + Err(error) => return Err(error), + } + } + }) + .await??; + assert!( + manager + .loaded + .lock() + .await + .contains_key(&evicted.repository_id) + ); + assert_eq!(manager.loaded.lock().await.len(), 3); + assert_eq!(manager.publication_budget.stats().foreground, 0); + server.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn production_shutdown_keeps_held_command_cell_heartbeat_and_workspace_until_producer_drains() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, files) = server().await?; + let manager = Arc::clone(&server.repositories); + let entry = create(&manager, "held", format).await?; + let held = held_renewal(&manager, &entry).await?; + let node = Arc::clone(&server.node); + let directory = server.directory.clone(); + let session: SessionId = server.advertisement.lock().await.advertisement().session(); + let mut shutdown = tokio::spawn(server.shutdown()); + timeout(Duration::from_secs(5), async { + while !manager.publication_budget.stats().closed { + tokio::task::yield_now().await; + } + }) + .await?; + assert!( + timeout(Duration::from_millis(30), &mut shutdown) + .await + .is_err() + ); + assert!(!node.is_shutting_down()); + assert!( + directory + .is_live(session, crate::server::unix_now_ms()?) + .await? + ); + assert!( + crate::server::workspace::Workspace::open(&files.path().join("node")) + .is_err_and(|error| error.kind() == std::io::ErrorKind::WouldBlock) + ); + assert!(matches!(held.state(), PublicationState::Held)); + assert_eq!(manager.publication_budget.stats().foreground, 1); + held.discard_held().await?; + timeout(Duration::from_secs(10), shutdown).await???; + assert!(node.is_shutting_down()); + assert!( + !directory + .is_live(session, crate::server::unix_now_ms()?) + .await? + ); + assert_eq!(manager.publication_budget.stats().foreground, 0); + drop((manager, held)); + let _reopened = crate::server::workspace::Workspace::open(&files.path().join("node"))?; + } + Ok(()) +} + +#[tokio::test] +async fn production_shutdown_recovers_other_repositories_while_one_producer_holds_its_command() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let (server, _files) = server().await?; + let manager = Arc::clone(&server.repositories); + let entries = [ + create(&manager, "held-first", format).await?, + create(&manager, "recover-second", format).await?, + ]; + // Stop automatic discovery without cancelling an owned round. The + // test observes recovery performed by the actual shutdown owner. + manager.recovery_scans.close(); + let order: Vec<_> = manager.loaded.lock().await.keys().copied().collect(); + let first = entries + .iter() + .find(|entry| entry.repository_id == order[0]) + .unwrap(); + let second = entries + .iter() + .find(|entry| entry.repository_id == order[1]) + .unwrap(); + let held = held_renewal(&manager, first).await?; + let (repository, client, service) = loaded(&manager, second.repository_id).await?; + let uncertain = held_renewal(&manager, second).await?; + service.coordinator.fault_for_test(fault); + uncertain.activate().await?; + let PublicationState::Uncertain(error) = + timeout(Duration::from_secs(10), uncertain.wait()).await? + else { + return Err("fault did not preserve uncertainty".into()); + }; + let PublicationError::Preparation(cellule_runtime::InvocationError::Pending(original)) = + error.as_ref() + else { + return Err("exact original evidence absent".into()); + }; + let sequence = match client.resolve(original).await? { + cellule_runtime::Resolution::Committed(receipt) => Some(receipt.commit_sequence()), + cellule_runtime::Resolution::Absent => None, + other => return Err(format!("unexpected original resolution {other:?}").into()), + }; + assert_eq!(sequence.is_some(), fault != 1); + assert_eq!(manager.publication_budget.stats().foreground, 2); + let node = Arc::clone(&server.node); + let mut shutdown = tokio::spawn(server.shutdown()); + // wait() intentionally returns retained uncertainty immediately. + // Observe its later terminal state without driving recovery here. + let resolved = timeout(Duration::from_secs(10), async { + loop { + let state = uncertain.state(); + if matches!( + state, + PublicationState::Finished(_) | PublicationState::Discarded + ) { + return state; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await?; + let PublicationState::Finished(Ok(PublicationOutcome::Preparation(outcome))) = resolved + else { + return Err( + format!("shutdown did not recover the exact renewal: {resolved:?}").into(), + ); + }; + let saved = RegisteredCustody::load_latest(&client, &repository.target, [81; 16]) + .await? + .ok_or("renewal registration absent")?; + assert_eq!(saved.evidence(), original.as_ref()); + assert_eq!(outcome.committed, saved.recover_preparation(&client).await?); + if let Some(sequence) = sequence { + assert_eq!(outcome.committed.receipt.commit_sequence, sequence); + } + assert!( + timeout(Duration::from_millis(30), &mut shutdown) + .await + .is_err() + ); + assert!(!node.is_shutting_down()); + assert!(matches!(held.state(), PublicationState::Held)); + assert_eq!(manager.publication_budget.stats().foreground, 1); + assert!(service.coordinator.stats().await.closed); + held.discard_held().await?; + timeout(Duration::from_secs(10), shutdown).await???; + assert!(node.is_shutting_down()); + assert_eq!(manager.publication_budget.stats().foreground, 0); + } + } + Ok(()) +} diff --git a/crates/canopy-server/src/server/residency/tests/serving.rs b/crates/canopy-server/src/server/residency/tests/serving.rs new file mode 100644 index 00000000..efa5345c --- /dev/null +++ b/crates/canopy-server/src/server/residency/tests/serving.rs @@ -0,0 +1,292 @@ +//! Exercise the actual manager, shared pool and server shutdown, not fixture wiring. +use super::recovery::{create, loaded, server}; +use super::*; +use crate::{ObjectFormat, ObjectId}; +use cellule_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; +use tokio::time::{Duration, timeout}; +type Result = std::result::Result>; + +async fn retained(repository: &RepositoryCell) -> Result { + let result = repository + .sql + .query( + None, + SqlBatch { + statements: vec![SqlStatement { + sql: "SELECT count(*) FROM catalog_serving_pins".into(), + parameters: vec![], + }], + }, + ) + .await?; + let Some([SqlValue::Integer(count)]) = result + .output + .first() + .and_then(|set| set.rows.first()) + .map(Vec::as_slice) + else { + return Err("serving count absent".into()); + }; + Ok(*count) +} +fn missing(format: ObjectFormat) -> ObjectId { + match format { + ObjectFormat::Sha1 => ObjectId::Sha1([7; 20]), + ObjectFormat::Sha256 => ObjectId::Sha256([7; 32]), + } +} + +#[tokio::test] +async fn production_browser_refs_use_certified_joint_roots_and_ignore_legacy_ref_tables() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let manager = server.repositories.clone(); + let entry = create(&manager, "browser-refs", format).await?; + let (repository, _, _) = loaded(&manager, entry.repository_id).await?; + // Trusted divergence isolates consumer authority. This old row must not + // become visible without publication of the immutable joint ref root. + repository.sql.batch(crate::server::mutation_identity()?, SqlBatch { + statements: vec![ + SqlStatement { sql: "UPDATE ref_generation SET default_branch='refs/heads/legacy',generation=99 WHERE singleton=1".into(), parameters: vec![] }, + SqlStatement { sql: "INSERT INTO refs(name,oid,version) VALUES('refs/heads/legacy',?1,1)".into(), parameters: vec![SqlValue::Blob(missing(format).to_vec())] }, + ], + }).await?; + let client = reqwest::Client::new(); + let url = format!( + "http://{}/api/repositories/browser-refs/browse", + server.address + ); + for query in [ + serde_json::json!({"kind":"resolve"}), + serde_json::json!({"kind":"refs"}), + ] { + let response = client + .post(&url) + .bearer_auth("local-recovery-test") + .json(&serde_json::json!({ + "repository_id": uuid::Uuid::from_bytes(entry.repository_id).to_string(), "query":query, + })) + .send() + .await?; + assert_eq!(response.status(), reqwest::StatusCode::OK); + let body: serde_json::Value = response.json().await?; + if query["kind"] == "resolve" { + assert_eq!(body["view"]["resolved"]["reference"], "refs/heads/main"); + assert!(body["view"]["resolved"]["oid"].is_null()); + assert!(body["view"]["resolved"]["version"].is_null()); + assert_eq!(body["view"]["resolved"]["generation"], 0); + } else { + assert_eq!(body["view"]["refs"]["default_branch"], "refs/heads/main"); + assert_eq!(body["view"]["refs"]["entries"], serde_json::json!([])); + assert_eq!(body["view"]["refs"]["generation"], 0); + } + } + let changed = client.post(&url).bearer_auth("local-recovery-test").json(&serde_json::json!({ + "repository_id": uuid::Uuid::from_bytes(entry.repository_id).to_string(), "query":{"kind":"refs","generation":99}, + })).send().await?; + assert_eq!(changed.status(), reqwest::StatusCode::CONFLICT); + let long = format!("refs/heads/{}", "\"".repeat(65_000)); + let continued = client + .post(&url) + .bearer_auth("local-recovery-test") + .json(&serde_json::json!({ + "repository_id": uuid::Uuid::from_bytes(entry.repository_id).to_string(), + "query":{"kind":"refs","generation":0,"after":long}, + })) + .send() + .await?; + assert_eq!(continued.status(), reqwest::StatusCode::OK); + let oversized = format!("refs/heads/{}", "x".repeat(65_536)); + let invalid = client + .post(&url) + .bearer_auth("local-recovery-test") + .json(&serde_json::json!({ + "repository_id": uuid::Uuid::from_bytes(entry.repository_id).to_string(), + "query":{"kind":"refs","generation":0,"after":oversized}, + })) + .send() + .await?; + assert_eq!(invalid.status(), reqwest::StatusCode::UNPROCESSABLE_ENTITY); + let anonymous = client + .post(&url) + .json(&serde_json::json!({ + "repository_id": uuid::Uuid::from_bytes(entry.repository_id).to_string(), "query":{"kind":"resolve"}, + })) + .send() + .await?; + assert_ne!(anonymous.status(), reqwest::StatusCode::OK); + drop(repository); + timeout(Duration::from_secs(10), server.shutdown()).await??; + } + Ok(()) +} + +#[tokio::test] +async fn shutdown_refuses_unpublished_serving_constructor_and_joins_it_before_workspace_release() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, files) = server().await?; + let manager = server.repositories.clone(); + let (entered, observed) = tokio::sync::oneshot::channel(); + let (proceed, waiting) = tokio::sync::oneshot::channel(); + *manager.serving_construction_gate.lock().await = Some((entered, waiting)); + let work = manager.clone(); + let creating = tokio::spawn(async move { work.create("late", format).await }); + timeout(Duration::from_secs(8), observed).await??; + let repository = manager + .loaded + .lock() + .await + .values() + .find(|loaded| loaded.name == "late") + .ok_or("late resident absent")? + .repository + .clone(); + // This public capability must be unavailable until its lifecycle owner + // is registered in the manager's drain inventory. + let premature = repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await; + let unavailable = premature.is_err(); + drop(premature); + let mut shutdown = tokio::spawn(server.shutdown()); + timeout(Duration::from_secs(8), manager.serving_stop.cancelled()).await?; + assert!( + timeout(Duration::from_millis(50), &mut shutdown) + .await + .is_err() + ); + assert!(!manager.publication_budget.stats().closed); + assert!( + crate::server::workspace::Workspace::open(&files.path().join("node")) + .is_err_and(|error| error.kind() == std::io::ErrorKind::WouldBlock) + ); + proceed.send(()).map_err(|_| "construction disappeared")?; + let result = timeout(Duration::from_secs(8), creating).await??; + timeout(Duration::from_secs(10), shutdown).await???; + assert!( + unavailable, + "unpublished constructor exposed serving before registered ownership" + ); + assert!(matches!( + result, + Err(crate::server::ServerError::Runtime( + cellule_runtime::Error::CellDraining + )) + )); + assert!( + repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await + .is_err() + ); + } + Ok(()) +} + +#[tokio::test] +async fn production_resident_shares_generation_and_busy_eviction_resumes_before_exact_drain() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let entry = create(&server.repositories, "pooled", format).await?; + let (repository, _client, service) = + loaded(&server.repositories, entry.repository_id).await?; + let first = repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await?; + let second = repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await?; + assert_eq!(first.fact(), second.fact()); + assert_eq!(retained(&repository).await?, 1); + assert!(!service.quiesce().await); + assert!(!service.coordinator.stats().await.closed); + assert_eq!(first.headers(&[missing(format)]).await?, vec![None]); + assert!( + repository + .serving_snapshot(ReadIdentity::Anonymous) + .await + .is_err() + ); + drop((first, second)); + assert!(timeout(Duration::from_secs(8), service.quiesce()).await?); + assert_eq!(retained(&repository).await?, 0); + assert!(service.coordinator.stats().await.closed); + assert!(service.quiesce().await); // retry after a later Cell release refusal + assert!( + repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await + .is_err() + ); + server.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn production_shutdown_keeps_publication_cell_heartbeat_and_workspace_until_last_borrow() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, files) = server().await?; + let manager = server.repositories.clone(); + let entry = create(&manager, "borrowed", format).await?; + let (repository, _client, service) = loaded(&manager, entry.repository_id).await?; + let first = repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await?; + let clone = first.clone(); + let node = server.node.clone(); + let directory = server.directory.clone(); + let session = server.advertisement.lock().await.advertisement().session(); + let mut shutdown = tokio::spawn(server.shutdown()); + timeout(Duration::from_secs(8), async { + loop { + if repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await + .is_err() + { + break; + } + tokio::task::yield_now().await; + } + }) + .await?; + assert!(!manager.publication_budget.stats().closed); + assert_eq!(retained(&repository).await?, 1); + assert!(!node.is_shutting_down()); + assert!( + directory + .is_live(session, crate::server::unix_now_ms()?) + .await? + ); + assert!( + crate::server::workspace::Workspace::open(&files.path().join("node")) + .is_err_and(|error| error.kind() == std::io::ErrorKind::WouldBlock) + ); + drop(first); + assert!( + timeout(Duration::from_millis(50), &mut shutdown) + .await + .is_err() + ); + assert_eq!(retained(&repository).await?, 1); + drop(clone); + timeout(Duration::from_secs(10), shutdown).await???; + assert!(node.is_shutting_down()); + assert!(manager.publication_budget.stats().closed); + assert!(service.coordinator.stats().await.closed); + assert!( + !directory + .is_live(session, crate::server::unix_now_ms()?) + .await? + ); + timeout(Duration::from_secs(2), service.serving.close_and_drain()).await?; + } + Ok(()) +} + +mod browser; diff --git a/crates/canopy-server/src/server/residency/tests/serving/browser.rs b/crates/canopy-server/src/server/residency/tests/serving/browser.rs new file mode 100644 index 00000000..5cd449d0 --- /dev/null +++ b/crates/canopy-server/src/server/residency/tests/serving/browser.rs @@ -0,0 +1,621 @@ +//! Actual HTTP reads against native, physically verified packs. Only the joint +//! catalog fact and editorial pull records are installed by trusted test SQL; +//! this does not qualify the still-unconverted live pull/ref producers. +use super::*; +use crate::packs::{ + catalog::{ + CatalogSnapshot, StoredCatalog, + serving_fixture::{BrowseFixture, operation, prepare}, + }, + directory::snapshot::DirectorySnapshot, + ref_state::RefStateSnapshotRoot, +}; +use base64::{Engine, engine::general_purpose::URL_SAFE_NO_PAD}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_runtime::codec::{BoundedEncoder, WireValue}; +use serde_json::{Value, json}; + +async fn install( + repository: &RepositoryCell, + generation: i64, + catalog: StoredCatalog, + refs: RefStateSnapshotRoot, +) -> Result { + let mut e = BoundedEncoder::new(256)?; + catalog.encode(&mut e)?; + let catalog = e.finish(); + let mut e = BoundedEncoder::new(128)?; + refs.encode(&mut e)?; + repository.sql.batch(crate::server::mutation_identity()?, SqlBatch {statements: vec![ + SqlStatement {sql:"INSERT INTO catalog_generations(generation,catalog,certificate,refs) VALUES(?1,?2,?3,?4)".into(),parameters:vec![SqlValue::Integer(generation),SqlValue::Blob(catalog),SqlValue::Blob(vec![42;32]),SqlValue::Blob(e.finish())]}, + SqlStatement {sql:"UPDATE catalog_state SET generation=?1 WHERE singleton=1".into(),parameters:vec![SqlValue::Integer(generation)]}, + ]}).await?; + Ok(()) +} +async fn request( + server: &crate::server::RunningServer, + suffix: &str, + body: Value, +) -> Result { + Ok(reqwest::Client::new() + .post(format!( + "http://{}/api/repositories/native-browser/{suffix}", + server.address + )) + .bearer_auth("local-recovery-test") + .json(&body) + .send() + .await?) +} +async fn browse(server: &crate::server::RunningServer, id: &str, query: Value) -> Result { + let response = request(server, "browse", json!({"repository_id":id,"query":query})).await?; + let status = response.status(); + let body = response.text().await?; + assert_eq!(status, reqwest::StatusCode::OK, "{body}"); + Ok(serde_json::from_str::(&body)?["view"].clone()) +} +fn tree(commit: ObjectId, path: &[u8], after: Option) -> Value { + json!({"kind":"tree","commit":hex::encode(commit),"path_base64":URL_SAFE_NO_PAD.encode(path),"after":after}) +} +fn file(commit: ObjectId, path: &[u8]) -> Value { + json!({"kind":"file","commit":hex::encode(commit),"path_base64":URL_SAFE_NO_PAD.encode(path)}) +} +async fn fixture( + server: &crate::server::RunningServer, + format: ObjectFormat, +) -> Result<(Arc, BrowseFixture, String)> { + let entry = create(&server.repositories, "native-browser", format).await?; + let (repository, _, _) = loaded(&server.repositories, entry.repository_id).await?; + let native = prepare( + format, + server.repositories.external_store.clone(), + entry.repository_id, + 40, + ) + .await?; + install(&repository, 2, native.catalog, native.refs).await?; + Ok(( + repository, + native, + uuid::Uuid::from_bytes(entry.repository_id).to_string(), + )) +} + +#[tokio::test] +async fn production_native_browser_preserves_raw_paths_modes_pages_tags_and_ordered_history() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let (repository, native, id) = fixture(&server, format).await?; + let first = browse(&server, &id, tree(native.tag, b"", None)).await?; + assert_eq!(first["tree"]["commit"]["oid"], hex::encode(native.main)); + assert_eq!(first["tree"]["tree_oid"], hex::encode(native.tree)); + let mut entries = first["tree"]["entries"] + .as_array() + .ok_or("entries")? + .clone(); + assert_eq!(entries.len(), 32); + let second = browse( + &server, + &id, + tree(native.main, b"", Some(first["tree"]["next_after"].clone())), + ) + .await?; + assert!(second["tree"]["next_after"].is_null()); + entries.extend( + second["tree"]["entries"] + .as_array() + .ok_or("entries")? + .clone(), + ); + assert_eq!(entries.len(), 49); + let paths = entries + .iter() + .map(|e| -> Result> { + Ok(URL_SAFE_NO_PAD.decode(e["path_base64"].as_str().ok_or("path")?)?) + }) + .collect::>>()?; + assert!(paths.windows(2).all(|p| p[0] < p[1])); + assert!(paths.contains(&b"\xffname".to_vec())); + let raw = entries + .iter() + .find(|e| e["path_base64"] == URL_SAFE_NO_PAD.encode(b"\xffname")) + .ok_or("raw entry")?; + assert!(raw["name"].is_null()); + for (path, mode, body) in [ + (b"link".as_slice(), "120000", b"src/lib.rs".as_slice()), + (b"executable", "100755", b"#!/bin/sh\nexit 0\n"), + (b"\xffname", "100644", b"raw name\n"), + (b"literal[?]*", "100644", b"literal path\n"), + (b"binary", "100644", b"a\0b\xff"), + (b"src/lib.rs", "100644", b"pub fn original() {}\n"), + ] { + let output = browse(&server, &id, file(native.main, path)).await?; + assert_eq!(output["file"]["mode"], mode); + assert_eq!(output["file"]["size"], body.len()); + assert_eq!( + output["file"]["content_base64"], + URL_SAFE_NO_PAD.encode(body) + ); + } + let directory = browse(&server, &id, tree(native.main, b"src", None)).await?; + assert_eq!( + directory["tree"]["entries"].as_array().ok_or("src")?.len(), + 1 + ); + let large = browse(&server, &id, file(native.main, b"large")).await?; + assert_eq!(large["file"]["content_status"], "too_large"); + assert_eq!(large["file"]["size"], native.large.len()); + assert!(large["file"]["content_base64"].is_null()); + let link = browse(&server, &id, file(native.main, b"submodule")).await?; + assert_eq!(link["file"]["content_status"], "gitlink"); + assert!(link["file"]["size"].is_null()); + let first = browse( + &server, + &id, + json!({"kind":"history","commit":hex::encode(native.tag)}), + ) + .await?; + let first = &first["history"]; + assert_eq!(first["commits"].as_array().ok_or("history")?.len(), 32); + assert_eq!( + first["commits"][0]["parents"], + json!([hex::encode(native.previous), hex::encode(native.side)]) + ); + let second = browse( + &server, + &id, + json!({"kind":"history","commit":first["next_commit"]}), + ) + .await?; + assert_eq!( + second["history"]["commits"] + .as_array() + .ok_or("continued history")? + .len(), + 9 + ); + assert!(second["history"]["next_commit"].is_null()); + let actual: Vec<_> = first["commits"] + .as_array() + .ok_or("first")? + .iter() + .chain(second["history"]["commits"].as_array().ok_or("second")?) + .map(|c| c["oid"].as_str().unwrap_or_default().to_owned()) + .collect(); + assert_eq!(actual, native.history); + // These tables do not exist: success cannot come from a fallback. + let tables=repository.sql.query(None,SqlBatch{statements:vec![SqlStatement{sql:"SELECT name FROM sqlite_schema WHERE name IN ('objects','object_closure','object_edges','commit_parents')".into(),parameters:vec![]}]}).await?; + assert!(tables.output[0].rows.is_empty()); + drop(repository); + timeout(Duration::from_secs(15), server.shutdown()).await??; + } + Ok(()) +} + +#[tokio::test] +async fn production_native_browser_refuses_absent_generations_wrong_formats_and_revoked_cached_access() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let (repository, native, id) = fixture(&server, format).await?; + let old = repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await?; + assert!(old.body(native.main, 1 << 20).await?.is_some()); + let _ = browse(&server, &id, tree(native.main, b"", None)).await?; + for (revision, status) in [ + (missing(format), reqwest::StatusCode::NOT_FOUND), + ( + match format { + ObjectFormat::Sha1 => ObjectId::Sha1([0; 20]), + ObjectFormat::Sha256 => ObjectId::Sha256([0; 32]), + }, + reqwest::StatusCode::NOT_FOUND, + ), + ( + match format { + ObjectFormat::Sha1 => ObjectId::Sha256([7; 32]), + ObjectFormat::Sha256 => ObjectId::Sha1([7; 20]), + }, + reqwest::StatusCode::UNPROCESSABLE_ENTITY, + ), + ] { + assert_eq!( + request( + &server, + "browse", + json!({"repository_id":id,"query":tree(revision,b"",None)}) + ) + .await? + .status(), + status + ); + } + let store = ArtifactStore::new( + server.repositories.external_store.clone(), + repository.repository_id(), + ); + let directory = DirectorySnapshot::empty(repository.repository_id(), format) + .upload(&store, operation(200)) + .await?; + let empty = CatalogSnapshot { + directory, + sources: None, + } + .upload(&store, operation(201)) + .await?; + install(&repository, 3, empty, native.refs).await?; + assert_eq!( + request( + &server, + "browse", + json!({"repository_id":id,"query":file(native.main,b"file-0000")}) + ) + .await? + .status(), + reqwest::StatusCode::NOT_FOUND + ); + assert!(old.body(native.main, 1 << 20).await?.is_some()); + drop(old); + // The cached pack must also obey current public/private authorization. + repository + .sql + .batch( + crate::server::mutation_identity()?, + SqlBatch { + statements: vec![SqlStatement { + sql: "UPDATE ref_generation SET visibility='public' WHERE singleton=1" + .into(), + parameters: vec![], + }], + }, + ) + .await?; + let public = repository.serving_snapshot(ReadIdentity::Anonymous).await?; + assert!(public.body(native.main, 1 << 20).await?.is_none()); + repository + .sql + .batch( + crate::server::mutation_identity()?, + SqlBatch { + statements: vec![SqlStatement { + sql: "UPDATE ref_generation SET visibility='private' WHERE singleton=1" + .into(), + parameters: vec![], + }], + }, + ) + .await?; + assert!(public.body(native.main, 1 << 20).await.is_err()); + let anonymous = reqwest::Client::new() + .post(format!( + "http://{}/api/repositories/native-browser/browse", + server.address + )) + .json(&json!({"repository_id":id,"query":tree(native.main,b"",None)})) + .send() + .await?; + assert_ne!(anonymous.status(), reqwest::StatusCode::OK); + drop((public, repository)); + timeout(Duration::from_secs(15), server.shutdown()).await??; + } + Ok(()) +} + +async fn editorial_pull( + repository: &RepositoryCell, + number: i64, + source: ObjectId, + base: ObjectId, +) -> Result { + // Only editorial metadata remains on legacy refs. The compared bodies and + // ancestry must come from the certified catalog, never objects/parents SQL. + let source_ref = format!("refs/heads/source-{number}"); + let base_ref = format!("refs/heads/base-{number}"); + repository.sql.batch(crate::server::mutation_identity()?,SqlBatch{statements:vec![ + SqlStatement{sql:"INSERT INTO refs(name,oid,version) VALUES(?1,?2,1),(?3,?4,1)".into(),parameters:vec![SqlValue::Text(source_ref.clone()),SqlValue::Blob(source.to_vec()),SqlValue::Text(base_ref.clone()),SqlValue::Blob(base.to_vec())]}, + SqlStatement{sql:"INSERT INTO pull_requests(number,id,creation_digest,author,title,body,state,draft,version,source_ref,base_ref,initial_source_oid,initial_base_oid,created_ms,updated_ms) VALUES(?1,?2,?3,'canopy','native comparison','','open',0,1,?4,?5,?6,?7,0,0)".into(),parameters:vec![SqlValue::Integer(number),SqlValue::Blob(uuid::Uuid::new_v4().into_bytes().to_vec()),SqlValue::Blob(vec![42;32]),SqlValue::Text(source_ref),SqlValue::Text(base_ref),SqlValue::Blob(source.to_vec()),SqlValue::Blob(base.to_vec())]}, + ]}).await?; + Ok( + json!({"kind":"current","revision":{"pull_version":1,"source_oid":hex::encode(source),"source_version":1,"base_oid":hex::encode(base),"base_version":1}}), + ) +} +#[tokio::test] +async fn production_native_comparisons_read_certified_ancestry_patches_and_previews() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let (repository, native, id) = fixture(&server, format).await?; + for (number, source, base, expected) in [ + (1, native.previous, native.side, native.root), + (2, native.main, native.side, native.side), + (3, native.main, native.main, native.main), + ] { + let target = editorial_pull(&repository, number, source, base).await?; + let response = request( + &server, + &format!("pulls/{number}/comparison"), + json!({"repository_id":id,"target":target,"query":{"kind":"files"}}), + ) + .await?; + let status = response.status(); + let body = response.text().await?; + assert_eq!(status, reqwest::StatusCode::OK, "{body}"); + let body: Value = serde_json::from_str(&body)?; + assert_eq!(body["comparison"]["merge_base"], hex::encode(expected)); + if number == 3 { + assert_eq!(body["comparison"]["files"], json!([])); + continue; + } + assert_eq!( + body["comparison"]["files"] + .as_array() + .ok_or("changes")? + .len(), + 1 + ); + assert_eq!(body["comparison"]["files"][0]["path"], "file-0000"); + for query in [ + json!({"kind":"patch","path_base64":URL_SAFE_NO_PAD.encode(b"file-0000")}), + json!({"kind":"file","path_base64":URL_SAFE_NO_PAD.encode(b"file-0000"),"side":"after"}), + ] { + let response = request( + &server, + &format!("pulls/{number}/comparison"), + json!({"repository_id":id,"target":target,"query":query}), + ) + .await?; + let status = response.status(); + let body = response.text().await?; + assert_eq!(status, reqwest::StatusCode::OK, "{body}"); + let body: Value = serde_json::from_str(&body)?; + if query["kind"] == "patch" { + assert_eq!(body["patch"]["status"], "text"); + assert_eq!(body["patch"]["merge_base"], hex::encode(expected)); + assert_eq!(body["patch"]["hunks"][0]["lines"][0]["kind"], "delete"); + assert_eq!(body["patch"]["hunks"][0]["lines"][1]["text"], "version 40"); + } else { + assert_eq!( + body["file"]["content_base64"], + URL_SAFE_NO_PAD.encode(b"version 40\n") + ); + } + } + } + // Equal non-commit tips must be refused even when merge_base can take + // its equal-input fast path. A catalog membership lookup is mandatory. + let blob = native.edges[&native.tree] + .iter() + .find(|edge| edge.expected_kind == crate::ObjectKind::Blob) + .ok_or("blob edge")? + .child; + let target = editorial_pull(&repository, 4, blob, blob).await?; + assert_eq!( + request( + &server, + "pulls/4/comparison", + json!({"repository_id":id,"target":target,"query":{"kind":"files"}}) + ) + .await? + .status(), + reqwest::StatusCode::SERVICE_UNAVAILABLE + ); + let target = editorial_pull(&repository, 5, missing(format), missing(format)).await?; + assert_eq!( + request( + &server, + "pulls/5/comparison", + json!({"repository_id":id,"target":target,"query":{"kind":"files"}}) + ) + .await? + .status(), + reqwest::StatusCode::NOT_FOUND + ); + drop(repository); + timeout(Duration::from_secs(15), server.shutdown()).await??; + } + Ok(()) +} + +#[tokio::test] +async fn production_certified_edge_pages_cover_wide_trees_and_parent_boundaries() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let entry = create(&server.repositories, "native-browser", format).await?; + let (repository, _, _) = loaded(&server.repositories, entry.repository_id).await?; + let native = prepare( + format, + server.repositories.external_store.clone(), + entry.repository_id, + 600, + ) + .await?; + install(&repository, 2, native.catalog, native.refs).await?; + let snapshot = repository + .serving_snapshot(ReadIdentity::Account("canopy")) + .await?; + let root_tree = native.edges[&native.root] + .iter() + .find(|edge| edge.expected_kind == crate::ObjectKind::Tree) + .ok_or("root tree")? + .child; + let wide = native.wide.ok_or("wide merge")?; + let mut ids = vec![ + native.main, + native.side, + wide, + native.tree, + root_tree, + missing(format), + ]; + ids.sort_unstable(); + let mut expected = Vec::new(); + for id in &ids { + if let Some(edges) = native.edges.get(id) { + let mut edges = edges.clone(); + edges.sort_by_key(|edge| edge.child); + edges.dedup_by_key(|edge| edge.child); + expected.extend(edges.into_iter().map(|edge| (*id, edge))); + } + } + assert!(expected.len() > 1024); + let mut cursor = None; + let mut actual = Vec::new(); + let mut absent = false; + let mut pages = 0; + loop { + let page = snapshot.edges_page(&ids, cursor).await?; + assert!(page.edges.len() <= 512); + assert!(page.headers.len() <= ids.len()); + absent |= page + .headers + .iter() + .any(|(id, header)| *id == missing(format) && header.is_none()); + actual.extend(page.edges); + pages += 1; + let Some(next) = page.next_after else { break }; + assert!(cursor.is_none_or(|old| old < next)); + cursor = Some(next); + } + assert!(pages >= 3); + assert!(absent); + assert_eq!(actual, expected); + let context = |error| { + matches!( + error, + Err(crate::packs::publication::ServingReadError::Context) + ) + }; + assert!(context(snapshot.edges_page(&[], None).await)); + assert!(context( + snapshot.edges_page(&[native.main; 129], None).await + )); + assert!(context( + snapshot.edges_page(&[native.main, native.main], None).await + )); + assert!(context( + snapshot + .edges_page( + &ids, + Some(( + match format { + ObjectFormat::Sha1 => ObjectId::Sha1([0; 20]), + ObjectFormat::Sha256 => ObjectId::Sha256([0; 32]), + }, + native.main + )) + ) + .await + )); + assert!(context( + snapshot + .edges_page( + &ids, + Some(( + native.main, + match format { + ObjectFormat::Sha1 => ObjectId::Sha1([0; 20]), + ObjectFormat::Sha256 => ObjectId::Sha256([0; 32]), + } + )) + ) + .await + )); + let wrong = match format { + ObjectFormat::Sha1 => ObjectId::Sha256([9; 32]), + ObjectFormat::Sha256 => ObjectId::Sha1([9; 20]), + }; + assert!(context(snapshot.edges_page(&[wrong], None).await)); + assert!(context( + snapshot.edges_page(&ids, Some((native.main, wrong))).await + )); + let mut parents = native.edges[&wide].clone(); + parents.sort_by_key(|edge| edge.child); + let (ordinal, last) = parents + .iter() + .enumerate() + .rev() + .find(|(_, edge)| { + edge.expected_kind == crate::ObjectKind::Commit && edge.child != native.root + }) + .ok_or("last parent")?; + assert!(ordinal >= 512); + let target = editorial_pull(&repository, 1, wide, last.child).await?; + let response = request(&server,"pulls/1/comparison",json!({"repository_id":uuid::Uuid::from_bytes(entry.repository_id).to_string(),"target":target,"query":{"kind":"files"}})).await?; + let status = response.status(); + let body = response.text().await?; + assert_eq!(status, reqwest::StatusCode::OK, "{body}"); + let body: Value = serde_json::from_str(&body)?; + assert_eq!(body["comparison"]["merge_base"], hex::encode(last.child)); + drop((snapshot, repository)); + timeout(Duration::from_secs(15), server.shutdown()).await??; + } + Ok(()) +} + +#[tokio::test] +async fn certified_http_clone_and_discovery_use_joint_refs_without_legacy_objects() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (server, _files) = server().await?; + let (repository, native, _) = fixture(&server, format).await?; + let work = tempfile::TempDir::new()?; + let url = format!("http://{}/canopy/native-browser.git", server.address); + let args = [ + "-c", + "http.extraHeader=Authorization: Bearer local-recovery-test", + "clone", + "--bare", + &url, + "clone.git", + ]; + crate::packs::catalog::serving_fixture::run_git(work.path(), &args, None) + .await + .map_err(|e| e.to_string())?; + let path = work.path().join("clone.git"); + let git = |args: Vec| { + let path = path.clone(); + async move { + let refs: Vec<_> = args.iter().map(String::as_str).collect(); + crate::packs::catalog::serving_fixture::run_git(&path, &refs, None) + .await + .map_err(|e| e.to_string()) + } + }; + assert_eq!( + String::from_utf8(git(vec!["rev-parse".into(), "HEAD".into()]).await?)?.trim(), + hex::encode(native.main) + ); + assert_eq!( + String::from_utf8( + git(vec!["rev-list".into(), "--count".into(), "HEAD".into()]).await? + )? + .trim(), + "42" + ); + assert_eq!( + git(vec!["show".into(), "HEAD:src/lib.rs".into()]).await?, + b"pub fn original() {}\n" + ); + git(vec!["fsck".into(), "--full".into()]).await?; + // Guessing a physically present, certified but unreferenced tag must be + // rejected by the HTTP gateway before native upload-pack sees the want. + let line = format!("want {}\n", hex::encode(native.tag)); + let body = format!("{:04x}{line}00000009done\n", line.len() + 4); + let refused = reqwest::Client::new() + .post(format!("{url}/git-upload-pack")) + .bearer_auth("local-recovery-test") + .header("Content-Type", "application/x-git-upload-pack-request") + .body(body) + .send() + .await?; + assert_eq!( + refused.status(), + reqwest::StatusCode::BAD_REQUEST, + "{}", + refused.text().await? + ); + drop(repository); + server.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/server/workspace/mod.rs b/crates/canopy-server/src/server/workspace/mod.rs index dafa4602..e7f1ab05 100644 --- a/crates/canopy-server/src/server/workspace/mod.rs +++ b/crates/canopy-server/src/server/workspace/mod.rs @@ -8,7 +8,7 @@ use std::{ }; const MARKER: &str = ".canopy-runtime"; -const FORMAT: &[u8] = b"canopy-runtime-v1\n"; +const FORMAT: &[u8] = crate::deployment::STORAGE_FORMAT.as_bytes(); struct OwnerLock(File); @@ -42,7 +42,17 @@ impl Workspace { fs::create_dir_all(directory)?; let directory = fs::canonicalize(directory)?; let owner = OwnerLock::acquire(&directory.join(".canopy-owner.lock"))?; - let root = directory.join("runtime-v1"); + match fs::symlink_metadata(directory.join("runtime-v1")) { + Ok(_) => { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + "legacy Canopy runtime requires a fresh packed-format data directory", + )); + } + Err(error) if error.kind() == io::ErrorKind::NotFound => {} + Err(error) => return Err(error), + } + let root = directory.join(crate::deployment::STORAGE_FORMAT); #[cfg(unix)] let created = { use std::os::unix::fs::DirBuilderExt; diff --git a/crates/canopy-server/src/server/workspace/tests.rs b/crates/canopy-server/src/server/workspace/tests.rs index 6d23bc68..0235936d 100644 --- a/crates/canopy-server/src/server/workspace/tests.rs +++ b/crates/canopy-server/src/server/workspace/tests.rs @@ -2,6 +2,46 @@ use super::*; type Result = std::result::Result>; +#[test] +fn legacy_runtime_prevents_cutover_and_preserves_every_file() -> Result { + let directory = tempfile::TempDir::new()?; + let legacy = directory.path().join("runtime-v1"); + fs::create_dir(&legacy)?; + fs::write(legacy.join(MARKER), b"canopy-runtime-v1\n")?; + fs::write(legacy.join("important"), b"retain old state")?; + assert!( + Workspace::open(directory.path()) + .is_err_and(|error| error.kind() == io::ErrorKind::InvalidData) + ); + assert_eq!(fs::read(legacy.join("important"))?, b"retain old state"); + assert!(!directory.path().join("canopy-pack-v1").exists()); + Ok(()) +} + +#[test] +fn packed_workspace_checks_format_before_reclaiming_any_file() -> Result { + for marker in [ + None, + Some(b"canopy-runtime-v1\n".as_slice()), + Some(b"unknown-format"), + ] { + let directory = tempfile::TempDir::new()?; + let root = directory.path().join("canopy-pack-v1"); + fs::create_dir(&root)?; + fs::write(root.join("important"), b"retain unrecognized state")?; + if let Some(marker) = marker { + fs::write(root.join(MARKER), marker)?; + } + assert!(Workspace::open(directory.path()).is_err()); + assert_eq!( + fs::read(root.join("important"))?, + b"retain unrecognized state" + ); + assert!(!directory.path().join("runtime-v1").exists()); + } + Ok(()) +} + #[test] fn releasing_ownership_unlocks_descriptors_retained_by_an_unrelated_child() -> Result { let directory = tempfile::TempDir::new()?; @@ -58,7 +98,7 @@ fn live_owner_blocks_cleanup_and_restart_reclaims_only_managed_state() -> Result #[test] fn unrecognized_runtime_is_never_reclaimed() -> Result { let directory = tempfile::TempDir::new()?; - let root = directory.path().join("runtime-v1"); + let root = directory.path().join(crate::deployment::STORAGE_FORMAT); fs::create_dir(&root)?; fs::write(root.join("important"), b"retain")?; assert!(Workspace::open(directory.path()).is_err()); @@ -78,9 +118,12 @@ fn runtime_symlink_is_rejected_and_nested_symlinks_do_not_delete_targets() -> Re let directory = tempfile::TempDir::new()?; let outside = tempfile::TempDir::new()?; fs::write(outside.path().join("important"), b"retain")?; - symlink(outside.path(), directory.path().join("runtime-v1"))?; + symlink( + outside.path(), + directory.path().join(crate::deployment::STORAGE_FORMAT), + )?; assert!(Workspace::open(directory.path()).is_err()); - fs::remove_file(directory.path().join("runtime-v1"))?; + fs::remove_file(directory.path().join(crate::deployment::STORAGE_FORMAT))?; let workspace = Workspace::open(directory.path())?; symlink(outside.path(), workspace.path().join("nested"))?; drop(workspace); diff --git a/crates/canopy-server/tests/multi_server/peers/cold_activation.rs b/crates/canopy-server/tests/multi_server/peers/cold_activation.rs index b6f66344..7350e360 100644 --- a/crates/canopy-server/tests/multi_server/peers/cold_activation.rs +++ b/crates/canopy-server/tests/multi_server/peers/cold_activation.rs @@ -236,7 +236,10 @@ async fn qualify_competing_cold_gateways(delay_claim_reply: bool) -> Result { .load(target.cell_id()) .await? .ok_or("missing control")?; - let local_path = format!("runtime-v1/{}/repository.sqlite", repository_id.simple()); + let local_path = format!( + "canopy-pack-v1/{}/repository.sqlite", + repository_id.simple() + ); let local_owners = ["first", "second"] .into_iter() .filter(|name| files.path().join(name).join(&local_path).exists()) diff --git a/crates/canopy-server/tests/multi_server/peers/mod.rs b/crates/canopy-server/tests/multi_server/peers/mod.rs index 479860fb..b6041dfc 100644 --- a/crates/canopy-server/tests/multi_server/peers/mod.rs +++ b/crates/canopy-server/tests/multi_server/peers/mod.rs @@ -173,7 +173,7 @@ async fn two_live_nodes_route_git_to_distinct_cell_owners_and_recover_the_direct let node = if name == "left" { "first" } else { "second" }; let other = if node == "first" { "second" } else { "first" }; let path = format!( - "runtime-v1/{}/repository.sqlite", + "canopy-pack-v1/{}/repository.sqlite", hex::encode(id.as_bytes()) ); assert!(files.path().join(node).join(&path).exists()); diff --git a/crates/canopy-server/tests/multi_server/residency/faults/mod.rs b/crates/canopy-server/tests/multi_server/residency/faults/mod.rs index 0b2db138..58ae4ba4 100644 --- a/crates/canopy-server/tests/multi_server/residency/faults/mod.rs +++ b/crates/canopy-server/tests/multi_server/residency/faults/mod.rs @@ -210,7 +210,7 @@ impl Fixture { let oid = run_git(Some(&source), &["rev-parse", "HEAD"]).await?; create(&client, address, "second").await?; create(&client, address, "third").await?; - let local = workspace.path().join("server/runtime-v1"); + let local = workspace.path().join("server/canopy-pack-v1"); Ok(Self { workspace, store, diff --git a/crates/canopy-server/tests/multi_server/residency/mod.rs b/crates/canopy-server/tests/multi_server/residency/mod.rs index 43862bb4..58068f1b 100644 --- a/crates/canopy-server/tests/multi_server/residency/mod.rs +++ b/crates/canopy-server/tests/multi_server/residency/mod.rs @@ -30,7 +30,7 @@ async fn repositories_beyond_resident_capacity_restore_git_and_lfs_on_the_same_n ); let authority = CellAuthority::new(layout); let server = CanopyServer::start(settings, store).await?; - let local_root = workspace.path().join("server/runtime-v1"); + let local_root = workspace.path().join("server/canopy-pack-v1"); let local = workspace.path().join("source"); run_git(None, &["init", "-b", "main", path_str(&local)?]).await?; run_git(Some(&local), &["config", "user.name", "Canopy Test"]).await?; diff --git a/crates/canopy-server/tests/multi_server/retained_catalog.rs b/crates/canopy-server/tests/multi_server/retained_catalog.rs index 88ab6ae3..c0e554b3 100644 --- a/crates/canopy-server/tests/multi_server/retained_catalog.rs +++ b/crates/canopy-server/tests/multi_server/retained_catalog.rs @@ -58,7 +58,7 @@ async fn retained_fixture() -> Result { .store_prefix .clone() .join("canopy-root-v1.json"), - Bytes::from_static(br#"{"kind":"service"}"#), + Bytes::from_static(br#"{"format":"canopy-pack-v1","purpose":{"kind":"service"}}"#), ) .await?; ApplicationIdentityStore::new(storage.clone(), configuration.store_prefix.clone()) diff --git a/crates/canopy-server/tests/multi_server/size.rs b/crates/canopy-server/tests/multi_server/size.rs index a3b9548b..0b01c4bd 100644 --- a/crates/canopy-server/tests/multi_server/size.rs +++ b/crates/canopy-server/tests/multi_server/size.rs @@ -99,7 +99,7 @@ async fn push_and_database_exceed_512_mib_and_lfs_exceeds_5_gib_after_restore() run_git(Some(&source), &["-c", AUTH, "push", &url, "main"]).await?; let expected = run_git(Some(&source), &["rev-parse", "HEAD"]).await?; tokio::fs::remove_dir_all(&source).await?; - let database_root = workspace.path().join("first/runtime-v1"); + let database_root = workspace.path().join("first/canopy-pack-v1"); let database_bytes = tokio::task::spawn_blocking( move || -> std::result::Result> { for entry in std::fs::read_dir(database_root)? { diff --git a/crates/canopy-server/tests/multi_server/ssh/publication.rs b/crates/canopy-server/tests/multi_server/ssh/publication.rs index 1910a6a2..aa9c8697 100644 --- a/crates/canopy-server/tests/multi_server/ssh/publication.rs +++ b/crates/canopy-server/tests/multi_server/ssh/publication.rs @@ -20,10 +20,16 @@ async fn fixture() -> Result { ssh_key::private::Ed25519Keypair::from_seed(&[12; 32]).into(), "test", )?; - let address = available_address().await?; - let server = CanopyServer::start( + let listener = TcpListener::bind("127.0.0.1:0").await?; + let address = listener.local_addr()?; + // Retain the advertised port across asynchronous startup; another test + // must not be able to claim it between address selection and serving. + let competition = TcpListener::bind(address).await.unwrap_err(); + assert_eq!(competition.kind(), std::io::ErrorKind::AddrInUse); + let server = CanopyServer::start_with_listener( server_config(address, workspace.path().join("server"), &host)?, store.clone(), + listener, ) .await?; create_repository(address, "publication").await?; @@ -192,10 +198,12 @@ async fn late_ssh_push_refusals_report_both_refs_and_survive_restore() -> Result assert_eq!(generation(&client, address).await?, before); server.shutdown().await?; - let address = available_address().await?; - let restored = CanopyServer::start( + let listener = TcpListener::bind("127.0.0.1:0").await?; + let address = listener.local_addr()?; + let restored = CanopyServer::start_with_listener( server_config(address, workspace.path().join("restored"), &host)?, store, + listener, ) .await?; let ssh_address = restored.ssh_addr().ok_or("SSH listener missing")?; @@ -303,10 +311,12 @@ async fn disconnected_ssh_push_finishes_publication_before_shutdown_releases_cel store.proceed.notify_one(); tokio::time::timeout(Duration::from_secs(20), draining).await???; - let address = available_address().await?; - let restored = CanopyServer::start( + let listener = TcpListener::bind("127.0.0.1:0").await?; + let address = listener.local_addr()?; + let restored = CanopyServer::start_with_listener( server_config(address, workspace.path().join("restored"), &host)?, store, + listener, ) .await?; let ssh_address = restored.ssh_addr().ok_or("SSH listener missing")?; @@ -341,10 +351,12 @@ async fn cold_ssh_push_preparation_failure_reports_rejection_before_any_refs_cha } = fixture().await?; git(Some(&source), &ssh, &["push", &url, "main"]).await?; server.shutdown().await?; - let address = available_address().await?; - let restored = CanopyServer::start( + let listener = TcpListener::bind("127.0.0.1:0").await?; + let address = listener.local_addr()?; + let restored = CanopyServer::start_with_listener( server_config(address, workspace.path().join("cold"), &host)?, store.clone(), + listener, ) .await?; let ssh_address = restored.ssh_addr().ok_or("SSH listener missing")?; diff --git a/crates/canopy-server/tests/multi_server/workspace.rs b/crates/canopy-server/tests/multi_server/workspace.rs index ea203625..49e35db7 100644 --- a/crates/canopy-server/tests/multi_server/workspace.rs +++ b/crates/canopy-server/tests/multi_server/workspace.rs @@ -1,5 +1,52 @@ use super::*; +#[tokio::test(flavor = "multi_thread")] +async fn old_deployment_format_is_rejected_before_workspace_or_identity_writes() +-> Result<(), Box> { + for bytes in [ + br#"{"kind":"service"}"#.as_slice(), + br#"{"purpose":{"kind":"service"}}"#, + br#"{"format":"future-format","purpose":{"kind":"service"}}"#, + br#"{"format":"canopy-pack-v1","purpose":{"kind":"backup","source":"source","pin":"11111111-1111-4111-8111-111111111111","complete":true}}"#, + br#"{"format":"canopy-pack-v1","purpose":{"kind":"restore","source":"source","pin":"11111111-1111-4111-8111-111111111111","complete":false}}"#, + ] { + let files = tempfile::TempDir::new()?; + let data = files.path().join("node"); + let packed = data.join("canopy-pack-v1"); + std::fs::create_dir_all(&packed)?; + std::fs::write(packed.join(".canopy-runtime"), b"canopy-pack-v1")?; + std::fs::write(packed.join("important"), b"retain existing cache")?; + let configuration = config(available_address().await?, data.clone()); + let store: Arc = Arc::new(InMemory::new()); + let storage = cellule_store::Store::new(Arc::clone(&store)); + let root = configuration + .store_prefix + .clone() + .join("canopy-root-v1.json"); + storage + .create_strict(&root, bytes::Bytes::copy_from_slice(bytes)) + .await?; + let original = storage.get_with_etag(&root).await?; + let identities = cellule_runtime::cell::application::ApplicationIdentityStore::new( + storage.clone(), + configuration.store_prefix.clone(), + ); + if let Ok(server) = CanopyServer::start(configuration, store).await { + server.shutdown().await?; + return Err("unversioned deployment was admitted".into()); + } + assert!(identities.load().await?.is_none()); + assert_eq!(storage.get_with_etag(&root).await?, original); + assert_eq!( + std::fs::read(packed.join("important"))?, + b"retain existing cache" + ); + assert!(!data.join(".canopy-owner.lock").exists()); + assert!(!data.join("runtime-v1").exists()); + } + Ok(()) +} + #[tokio::test(flavor = "multi_thread")] async fn active_node_blocks_workspace_reuse_and_shutdown_allows_durable_restore() -> Result<(), Box> { @@ -27,7 +74,7 @@ async fn active_node_blocks_workspace_reuse_and_shutdown_allows_durable_restore( matches!(occupied, Err(canopy_server::server::ServerError::Io(error)) if error.kind() == std::io::ErrorKind::WouldBlock) ); server.shutdown().await?; - let sentinel = data.join("runtime-v1/abandoned"); + let sentinel = data.join("canopy-pack-v1/abandoned"); std::fs::write(&sentinel, b"reclaim before restoring")?; let address = available_address().await?; let server = CanopyServer::start(config(address, data), store).await?; @@ -44,3 +91,292 @@ async fn active_node_blocks_workspace_reuse_and_shutdown_allows_durable_restore( server.shutdown().await?; Ok(()) } + +#[tokio::test(flavor = "multi_thread")] +async fn new_repositories_bootstrap_the_production_packed_catalog_before_becoming_ready() +-> Result<(), Box> { + use canopy_server::packs::{ + catalog::{CatalogSnapshot, StoredCatalog}, + ref_state::RefStateSnapshotRoot, + }; + use cellule_runtime::codec::{BoundedDecoder, WireValue}; + use object_store::ObjectStoreExt; + for format in ["sha1", "sha256"] { + let files = tempfile::TempDir::new()?; + let data = files.path().join("node"); + let store: Arc = Arc::new(InMemory::new()); + let listener = TcpListener::bind("127.0.0.1:0").await?; + let address = listener.local_addr()?; + let cfg = config(address, data.clone()); + let prefix = cfg.store_prefix.clone(); + let tenant = cfg.tenant; + let application = cfg.application; + let layout = cellule_runtime::ltx::CellStorageLayout::new( + cellule_store::Store::new(Arc::clone(&store)), + prefix.clone(), + *application.as_bytes(), + ); + let server = CanopyServer::start_with_listener(cfg, Arc::clone(&store), listener).await?; + let client = reqwest::Client::new(); + let response: serde_json::Value = client + .post(format!("http://{address}/api/repositories")) + .bearer_auth("local-test-token") + .json(&serde_json::json!({"name":"packed","object_format":format})) + .send() + .await? + .error_for_status()? + .json() + .await?; + let repository = uuid::Uuid::parse_str( + response["repository_id"] + .as_str() + .ok_or("repository UUID absent")?, + )?; + let target = canopy_server::repository_target(tenant, application, *repository.as_bytes())?; + server.shutdown().await?; + let root = repository_root(&layout, &target, &files.path().join("original.sqlite")).await?; + let (catalog, refs, allocation, admission, custody) = { + let connection = root.connection()?; + for table in [ + "objects", + "object_uploads", + "object_chunks", + "object_edges", + "object_closure", + "git_packs", + "commit_parents", + "commit_ancestry", + ] { + assert!( + !connection.query_row( + "SELECT EXISTS(SELECT 1 FROM sqlite_master WHERE type='table' AND name=?1)", + [table], + |row| row.get::<_, bool>(0) + )?, + "legacy table {table} is still selected" + ); + } + let (generation,catalog,refs,certificate): (i64,Vec,Vec,Vec)=connection.query_row( + "SELECT g.generation,g.catalog,g.refs,g.certificate FROM catalog_state s JOIN catalog_generations g ON g.generation=s.generation WHERE s.singleton=1",[],|row| Ok((row.get(0)?,row.get(1)?,row.get(2)?,row.get(3)?)))?; + assert_eq!(generation, 1); + assert_eq!(certificate.len(), 32); + assert_eq!( + connection + .query_row("SELECT count(*) FROM catalog_initialization", [], |row| row + .get::<_, i64>(0))?, + 1 + ); + assert_eq!( + connection + .query_row("SELECT count(*) FROM catalog_operations", [], |row| row + .get::<_, i64>(0))?, + 0 + ); + assert_eq!(connection.query_row("SELECT count(*) FROM catalog_leases WHERE generation=0 AND recovery IS NOT NULL", [], |row| row.get::<_, i64>(0))?, 0); + assert_eq!( + connection.query_row( + "SELECT count(*) FROM catalog_recovery_receipts", + [], + |row| row.get::<_, i64>(0) + )?, + 1 + ); + let allocation = connection.query_row( + "SELECT artifact_sequence FROM repository_identity WHERE singleton=1", + [], + |row| row.get::<_, i64>(0), + )?; + assert_eq!(allocation, 1); + let admission: Vec = + connection.query_row("SELECT initial_preparation FROM pushes", [], |row| { + row.get(0) + })?; + assert!(!admission.is_empty()); + assert!(admission.len() <= 1024); + let custody: (Vec, Vec) = connection.query_row( + "SELECT intent,phase FROM catalog_custody_commands WHERE step=0", + [], + |row| Ok((row.get(0)?, row.get(1)?)), + )?; + assert!(custody.0.len() <= 4096 && custody.1.len() <= 1024); + assert_eq!( + connection.query_row( + "SELECT count(*) FROM catalog_custody_commands", + [], + |row| row.get::<_, i64>(0) + )?, + 1 + ); + (catalog, refs, allocation, admission, custody) + }; + drop(root); + let mut decoder = BoundedDecoder::new(&catalog, 256)?; + let catalog = StoredCatalog::decode(&mut decoder)?; + decoder.finish()?; + let mut decoder = BoundedDecoder::new(&refs, 128)?; + let refs = RefStateSnapshotRoot::decode(&mut decoder)?; + decoder.finish()?; + let provider: Arc = Arc::new(object_store::prefix::PrefixStore::new( + Arc::clone(&store), + prefix.clone(), + )); + let artifacts = canopy_object_storage::artifact::ArtifactStore::new( + Arc::clone(&provider), + *repository.as_bytes(), + ); + let snapshot = CatalogSnapshot::download(&artifacts, catalog).await?; + assert!(snapshot.sources.is_none()); + let snapshot = refs.read(&artifacts).await?; + assert_eq!(snapshot.generation, 0); + assert_eq!(snapshot.default_branch, "refs/heads/main"); + assert!(snapshot.root.is_none()); + let restored_data = files.path().join("restored"); + let listener = TcpListener::bind("127.0.0.1:0").await?; + let restored_address = listener.local_addr()?; + let restored = CanopyServer::start_with_listener( + config(restored_address, restored_data.clone()), + Arc::clone(&store), + listener, + ) + .await?; + let restored_response: serde_json::Value = client + .get(format!("http://{restored_address}/api/repositories/packed")) + .bearer_auth("local-test-token") + .send() + .await? + .error_for_status()? + .json() + .await?; + assert_eq!( + restored_response["repository_id"], + response["repository_id"] + ); + restored.shutdown().await?; + let root = repository_root(&layout, &target, &files.path().join("restored.sqlite")).await?; + { + let connection = root.connection()?; + assert_eq!( + connection.query_row( + "SELECT intent,phase FROM catalog_custody_commands WHERE step=0", + [], + |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, Vec>(1)?)) + )?, + custody + ); + assert_eq!( + connection.query_row( + "SELECT count(*) FROM catalog_custody_commands", + [], + |row| row.get::<_, i64>(0) + )?, + 1 + ); + assert_eq!( + connection.query_row("SELECT initial_preparation FROM pushes", [], |row| row + .get::<_, Vec>(0))?, + admission + ); + assert_eq!( + connection.query_row( + "SELECT artifact_sequence FROM repository_identity WHERE singleton=1", + [], + |row| row.get::<_, i64>(0) + )?, + allocation + ); + assert_eq!( + connection.query_row( + "SELECT catalog FROM catalog_generations WHERE generation=1", + [], + |row| row.get::<_, Vec>(0) + )?, + { + let mut encoded = cellule_runtime::codec::BoundedEncoder::new(256)?; + catalog.encode(&mut encoded)?; + encoded.finish() + } + ); + } + drop(root); + + // A Ready repository must fail closed when retained initialization + // metadata is missing, rather than silently publishing another empty + // catalog. The immutable Cell outcome still exists in this fixture. + let catalog_path = artifacts.path( + canopy_object_storage::artifact::ArtifactKey { + operation: catalog.operation, + binding_digest: catalog.artifact.digest, + kind: canopy_object_storage::artifact::ArtifactKind::CatalogNode, + }, + catalog.artifact.digest, + )?; + provider.head(&catalog_path).await?; + provider.delete(&catalog_path).await?; + assert!(matches!( + provider.head(&catalog_path).await, + Err(object_store::Error::NotFound { .. }) + )); + let listener = TcpListener::bind("127.0.0.1:0").await?; + let broken_address = listener.local_addr()?; + let broken = CanopyServer::start_with_listener( + config(broken_address, files.path().join("missing-initial-catalog")), + Arc::clone(&store), + listener, + ) + .await?; + let refused = client + .get(format!("http://{broken_address}/api/repositories/packed")) + .bearer_auth("local-test-token") + .send() + .await?; + assert_eq!(refused.status(), reqwest::StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(refused.text().await?, "Repository is unavailable"); + broken.shutdown().await?; + let root = repository_root(&layout, &target, &files.path().join("refused.sqlite")).await?; + { + let connection = root.connection()?; + assert_eq!( + connection.query_row( + "SELECT artifact_sequence FROM repository_identity WHERE singleton=1", + [], + |row| row.get::<_, i64>(0) + )?, + allocation + ); + assert_eq!( + connection.query_row( + "SELECT generation FROM catalog_state WHERE singleton=1", + [], + |row| row.get::<_, i64>(0) + )?, + 1 + ); + } + drop(root); + } + Ok(()) +} + +// Restored workers use sparse placeholders. Ordinary SQLite cannot fetch their +// missing pages; inspect the authenticated published root through Cellule's VFS. +async fn repository_root( + layout: &cellule_runtime::ltx::CellStorageLayout, + target: &cellule_runtime::CellTarget, + destination: &Path, +) -> Result> { + let control = cellule_runtime::control::authority::CellAuthority::new(layout.clone()) + .load(target.cell_id()) + .await? + .ok_or("repository control absent")?; + let root = control.value().ltx_root().ok_or("repository root absent")?; + let replica = cellule_ltx::CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *control.value().incarnation.as_bytes(), + cellule_ltx::Limits::default(), + )?; + Ok(replica + .open_root(&root) + .await? + .open_read_only(destination)?) +} diff --git a/docs/README.md b/docs/README.md index 71320f82..11df3ac2 100644 --- a/docs/README.md +++ b/docs/README.md @@ -16,6 +16,7 @@ Use this page to choose a document by task. Canopy's hosting core supports stock | Plan or evaluate capacity | [Repository density and latency](performance-plan.md) | Workloads, targets, measured results and limits of each result | | Recover an explicitly supported predecessor Cell contract | [Retained-contract maintenance recovery](performance/2026-10-02-retained-maintenance-recovery.md) | Recovery admission fix, regression scope and incomplete rebuild status | | Implement large-repository storage for a large team | [Packed storage design](large-repository-storage-design.md), [implementation plan](large-repository-implementation-plan.md) and [large-team amendment](large-team-scalability.md) | Hard-cutover design, required scalability changes and single-hot-repository release gates; capacity remains unqualified | +| Inspect certified browser and native transport reads | [Serving contract](design/certified-serving-pins.md), [implementation status](large-repository-implementation-status.md) and [workspace evidence](evidence/serving-certified-workspaces-20261004.json) | Owned local readers and native fetch workspaces, bounded bodies/edges, actual HTTP clones and remaining producer/remote/capacity gates | | Inspect the original RustFS corpus upgrade | [Full-corpus activation and remote verification](performance/2026-10-01-original-corpus-activation.md) | Admission of 10,003 Cells, full Git/LFS verification, failed diagnostic load windows and open owner-loss gates | | Track three nodes behind a proxy and the latest dependency candidate | [Three-node proxy qualification](performance/2026-09-30-three-node-proxy.md) | Baseline/candidate pins, complete-corpus recovery, failing load windows and open gates | | Inspect the merged workspace and RustFS verification | [Workspace and RustFS verification](performance/2026-09-30-workspace-rustfs.md) | The `70bd25f` revision, conflict resolution and end-to-end gates | diff --git a/docs/contracts.md b/docs/contracts.md index 1e20e1b2..6f0ca384 100644 --- a/docs/contracts.md +++ b/docs/contracts.md @@ -1134,7 +1134,7 @@ complete OOM/CPU/process/crash fault matrix require separate qualification. `data_dir/.canopy-owner.lock` prevents concurrent nodes from using one local directory. The server and repository manager retain the lock throughout their -lifetime, including detached request work. `data_dir/runtime-v1/` is private to +lifetime, including detached request work. `data_dir/canopy-pack-v1/` is private to the node (created with mode 0700 on Unix) and identified by a version marker. Startup validates that marker and reclaims local SQLite files, Git caches and spools before opening Cells or advertising the node. These are disposable copies; @@ -2647,8 +2647,14 @@ with the release still Ready provide a quiet capture window. A successful source pin UUID identifies one immutable cut; repeating create reuses it. `/canopy-root-v1.json` is a bounded, conditional-write reservation with -Service, Backup or Restore purpose. Backup/Restore bind the source prefix and -pin UUID. Service initialization competes on this same key. Every reservation, +the exact `{"format":"canopy-pack-v1","purpose":...}` envelope containing +the existing Service, Backup or Restore purpose. Missing/unknown formats and +unversioned purpose records are rejected, and the previous top-level-purpose +decoder rejects the new envelope. Backup/Restore bind the source prefix and +pin UUID. Startup checks format and serving eligibility before local workspace +reclamation, storage probes, identity/release writes and Cell activation. +Legacy `runtime-v1` directories and missing/unknown current workspace markers +are retained and rejected; operators select a fresh data directory. Service initialization competes on this same key. Every reservation, including Service, rejects an existing application identity without a root marker. Current initialization writes the marker first; admission rechecks it after reading an identity to allow a concurrent current-format creator. Copies also diff --git a/docs/delivery-plan.md b/docs/delivery-plan.md index bac79695..33549ee9 100644 --- a/docs/delivery-plan.md +++ b/docs/delivery-plan.md @@ -1822,7 +1822,7 @@ qualification remain open. ### Managed local runtime recovery qualification On 2026-09-26, `src/server/workspace.rs` replaced anonymous node directories with -one marked `runtime-v1/` directory beneath a locked `data_dir`. The manager retains +one marked `canopy-pack-v1/` directory beneath a locked `data_dir`. The manager retains that owner with detached request work. Git caches use a recognizable prefix and per-cache worker locks. On Unix, native Git and its descendants inherit the lock descriptor across exec. Startup acquires every abandoned worker fence before diff --git a/docs/design/bound-preparation-dispatch.md b/docs/design/bound-preparation-dispatch.md index 7a8a37c7..0bb8bfad 100644 --- a/docs/design/bound-preparation-dispatch.md +++ b/docs/design/bound-preparation-dispatch.md @@ -1,16 +1,16 @@ # Exact bound preparation command ownership -Long bound preparations and owner takeover need an exact recovery path for ClaimPreparation and RenewPreparation. ReadyPreparation::claim and PreparationSession::ready_renew prepare those SDK commands before admission to the existing PublicationCoordinator. They reuse LeaseCheck, LeaseRequest, PreparationToken, independent generation pins and the same coordinator job, actor queue and command evidence. The private request is boxed so its exact command/context does not enlarge every ReadyPublication value. No schema, command ID, outbox representation or compatibility adapter is added. +Long bound preparations and owner takeover use `ReadyPreparation::claim`, `PreparationSession::ready_renew` and `PreparationBaseResolver::ready_renew` to prepare original typed custody command 42 and exact registrar 41 before admission to the existing PublicationCoordinator. These factories reuse LeaseCheck, LeaseRequest, PreparationToken, independent generation pins and the same coordinator job, actor queue and command evidence. Direct caller-owned raw renewal APIs have been removed. The private request is boxed so its exact command/context does not enlarge every ReadyPublication value. The [custody journal](durable-custody-command-intents.md) is the original command's durable representation; there is no compatibility adapter for raw commands 12/13. ## Admission and recovery -Both commands enter the foreground class with an 8 KiB reservation for two bounded encoded copies. Account/operation limits, operation-ID exclusivity, FIFO account rotation, reserved maintenance slots and concurrent durability waits are shared with checkpoints and push completions. An uncertain command keeps its exact identity, bytes and credits. Dropping an observer does not cancel an accepted command. pending, recover and close_and_drain retain their existing behavior; recovery remains available after closing. Refused admission returns the original ready value. +Both commands enter the foreground class with a 28 KiB reservation covering retained/dispatch intents, registrar transport/query decode and original body/reply ceilings. Global byte, class/account, operation and durability-wait caps stay unchanged. Account/operation limits, operation-ID exclusivity, FIFO account rotation, reserved maintenance slots and concurrent durability waits are shared with checkpoints and push completions. An uncertain command keeps its exact identity, bytes and credits. Dropping an observer does not cancel an accepted command. pending, recover and close_and_drain retain their existing behavior; recovery remains available after closing. Refused admission returns the original ready value. -The factories check repository target equality, a nonzero lease duration within MAX_LEASE_MS and a 4 KiB encoded request. Renewal checks its shared session before and after SDK preparation. Claim intentionally accepts a previous-owner or expired token: the authoritative command checks exact operation identity, actor, phase, current write access, current admitted owner and SQL pin/quota invariants. An expired source may be claimed while it remains present; it cannot be renewed. Claim does not grant custody over old input bytes. Adopt and register the authenticated retained checkpoint before using borrowed physical inputs. +The factories check repository target equality, a nonzero lease duration within MAX_LEASE_MS and the complete 1 KiB original body/4 KiB intent limits. Renewal checks its shared session before and after SDK preparation. Claim intentionally accepts a previous-owner or expired token: the authoritative command checks exact operation identity, actor, phase, current write access, current admitted owner and SQL pin/quota invariants. An expired source may be claimed while it remains present; it cannot be renewed. Claim does not grant custody over old input bytes. Adopt and register the authenticated retained checkpoint before using borrowed physical inputs. ## Original outcome and fresh custody -PublicationOutcome::Preparation returns PreparationCommandOutcome with the command kind, original Committed and a separate session Result. Neither the reply's recorded timestamps nor replay alone constructs usable custody. Claim freshly opens CheckPreparation at the original receipt, matches the granted token, floor and format, and exposes a new private session. Renewal freshly queries the same attempt/floor/format and updates the existing shared conservative deadline. Existing fences remain permanent. Each query measures its local deadline from before request dispatch so queue/transport time shortens usable custody. +PublicationOutcome::Preparation returns PreparationCommandOutcome with the command kind, original Committed and a separate session Result. Neither the reply's recorded timestamps nor replay alone constructs usable custody. Claim freshly opens CheckPreparation at the original receipt, matches the granted token, floor and format, and exposes a new private session. Renewal freshly queries the same attempt/floor/format and updates the existing shared conservative deadline. Existing fences remain permanent. `restore_renewal` on a session or base resolver reconstructs the registered original while preserving that same shared fence and deadline; restoring a separate session would leave old resolvers usable after a failed fresh observation. Each query measures its local deadline from before request dispatch so queue/transport time shortens usable custody. A known success remains a known success when current permission, expiry or a later Claim prevents usable custody. The original receipt is returned alongside the custody error. Renewal failures fence the original shared session; an ambiguous command retains its old conservative deadline until resolved. Exact rejected/not-started renewals fence that session. Claim does not depend on or revive a previous local session. Final proof factories and authoritative commands continue to recheck custody independently. Preparation outcomes cannot become Git push responses. @@ -18,7 +18,7 @@ Renewal preserves the original generation floor and creating namespace. Claim cr ## Remaining lifecycle work -This dispatcher owns one accepted Claim or Renew command, not the entire bound preparation lifecycle. Automatic renewal, bound worker/result ownership, checkpoint serialization, a separate local residence ceiling and shutdown drain now reuse the staging lifecycle; see the [bound lifecycle contract](bound-preparation-lifecycle.md). Final-publication lifecycle serialization, production producer integration, SQL floor-capacity qualification and durable takeover reconstruction remain required. The process-local exact command survives caller cancellation, not process loss. Unknown or expired SDK evidence cannot justify issuing a replacement command; durable exact/logical recovery must preserve that distinction. Complete retained-root enumeration, writer/reader drain and isolated restore remain required before collection. No remote deletion authority is introduced. +This dispatcher owns one accepted Claim or Renew command, not the entire bound preparation lifecycle. Automatic renewal, bound worker/result ownership, checkpoint serialization, a separate local residence ceiling and shutdown drain now reuse the staging lifecycle; see the [bound lifecycle contract](bound-preparation-lifecycle.md). Final-publication lifecycle serialization, production producer integration, SQL floor-capacity qualification and durable takeover reconstruction remain required. `ReadyPreparation::restore` reconstructs the registered Claim/Renew head with its original identity and bytes; it submits through the same fair queue. Known outcome restoration precedes fresh session observation. Current actual-owner fencing must still be integrated with cold session construction: a successful SQL lease query alone does not prove the active owner epoch. That is an explicit release gate, not completed takeover authority. Unknown or expired SDK evidence cannot justify issuing a replacement command; durable exact/logical recovery must preserve that distinction. Complete retained-root enumeration, writer/reader drain and isolated restore remain required before collection. No remote deletion authority is introduced. ## Validation scope diff --git a/docs/design/bound-preparation-lifecycle.md b/docs/design/bound-preparation-lifecycle.md index 1eb28e56..667aac42 100644 --- a/docs/design/bound-preparation-lifecycle.md +++ b/docs/design/bound-preparation-lifecycle.md @@ -6,9 +6,9 @@ StagingCoordinator now keeps an operation admitted after BindStaging and can adm Seal drains staged workers/results before Bind as before. A known Bind or bound Claim stores its original lease and receipt in bound_result before fresh queries. Successful fresh CheckPreparation token/floor/format matching constructs the shared bound PreparationSession. Staging contexts become inactive at that phase transition. Admission remains charged until explicit stop, loss of custody or the residence ceiling; Bound is a recorded handoff result, not release of service ownership. -bound_session returns a locally live shared session only in the usable bound phase. open_base refreshes at the original bound receipt and constructs the existing PreparationBaseResolver with that session's deadline/fence. Base reads, reconciliation and private proof factories observe the same session. spawn_bound admits a callback into the existing global/actor worker slots and typed StagingTask handoff. Dropped callers retain execution/results in service ownership. Retrieved results transfer once; failed/expired results drop before their credits. Use the existing admitted native/disk/reader primitives inside callbacks; these worker counters do not account for arbitrary heap, unjoined descendants or detached I/O. +bound_session returns a locally live shared session only in the usable bound phase. open_base refreshes at the original bound receipt and constructs the existing PreparationBaseResolver with that session's deadline/fence. Base reads, reconciliation and private proof factories observe the same session. spawn_bound admits a callback receiving both the shared preparation session and a StagingContext into the existing global/actor worker slots and typed StagingTask handoff. Context clones and their internal physical_owner retain that original worker admission until every physical job drains, including after custody fencing or result transfer; lifetime ownership grants no new authority. Dropped callers retain execution/results in service ownership. Retrieved results transfer once; failed/expired results drop before their credits. Use the existing admitted native/disk/reader primitives inside callbacks; these worker counters do not account for arbitrary heap, unjoined descendants or detached I/O. -A public Bound result remains recoverable after graceful stop and failed fresh custody. It is not usable authority. bound_session/open_base reject stopped or fenced jobs. Previously opened bases and session clones observe the shared permanent fence and residence ceiling. +A public Bound result remains recoverable after graceful stop and failed fresh custody. It is not usable authority. bound_session/open_base reject stopped or fenced jobs. Previously opened bases and session clones observe the shared permanent fence and residence ceiling. The session also retains a terminal fence notification: in-flight bound callbacks wake without waiting for a renewal or a coordinator status change, and late subscribers see the existing fence. Cancellation aborts and joins the callback before resource credit returns. ## Renewal and residence @@ -20,9 +20,9 @@ The local ceiling does not shorten or remove the independent SQL pin. Already ad ## Adopted checkpoint and shutdown -A bound Claim may adopt its authenticated retained input root through the shared session. register_inputs now accepts that matching adopted certificate in the bound phase and uses the existing single 4 KiB checkpoint/result slot, adding it to the 8 KiB command-copy reservation. Due renewal precedes queued registration. RegisterStagedInputs shares the exact slot with renewal; its original receipt is stored before fresh checkpoint-digest and bound-session queries. Fresh custody failure fences the job while the committed registration remains observable through pending_inputs. The original creating namespace and input root are reused; no nodes or native pairs are copied. A staged checkpoint already consumes that operation's single slot. +A bound Claim may adopt its authenticated retained input root through the shared session. register_inputs now accepts that matching adopted certificate in the bound phase and uses the existing single 4 KiB checkpoint/result slot, adding it to the 28 KiB registered-custody command-wire reservation. Due renewal precedes queued registration. RegisterStagedInputs shares the exact slot with renewal; its original receipt is stored before fresh checkpoint-digest and bound-session queries. Fresh custody failure fences the job while the committed registration remains observable through pending_inputs. The original creating namespace and input root are reused; no nodes or native pairs are copied. A staged checkpoint already consumes that operation's single slot. -stop/close refuse new staged or bound workers, keep renewing while accepted tasks and retained results drain, and retain uncertain exact evidence/credits. Service consumers must retrieve completed results to finish graceful drain. Once drained, the shared bound session is fenced before operation admission is returned. Reached residence, worker error/panic or lost custody aborts and joins outstanding callbacks and discards untransferred results before credit release. SQL pins and remote artifacts remain independently retained. +stop/close refuse new staged or bound workers, keep renewing while accepted tasks and retained results drain, and retain uncertain exact evidence/credits. Service consumers must retrieve completed results to finish graceful drain. Once drained, the shared bound session is fenced before operation admission is returned. Reached residence, worker error/panic or lost custody aborts and joins outstanding async callbacks and discards untransferred results. Admission remains charged until their retained physical owners also drain. SQL pins and remote artifacts remain independently retained. ## Remaining integration and release gates diff --git a/docs/design/certified-serving-pins.md b/docs/design/certified-serving-pins.md new file mode 100644 index 00000000..1e2ff3c3 --- /dev/null +++ b/docs/design/certified-serving-pins.md @@ -0,0 +1,554 @@ +# Certified serving generations and physical drain + +Serving readers need an authorized immutable catalog/ref snapshot whose retained +artifacts cannot disappear while an owned worker is suspended. The implementation +adds a bounded serving-pin receiver and an owned metadata-read capability. This +is a foundation for production reader conversion. A service-owned producer now +acquires, retains, renews and drains one generation independently of its callers; +the production manager now creates a bounded resident pool and exposes its +borrow through `RepositoryCell::serving_snapshot`. Local browser and comparison object reads now use this capability. The native +workspace conversion also routes local HTTP/SSH discovery and fetch through +accepted joint refs and their certified forward graph. Remaining generated/write +producers, product consumers and remote routes still require conversion. +The branch remains unreleasable until that conversion and the full cutover gates +are complete. + +## Atomic selection and independent retention + +`AcquireServingPin` selects the current nonzero `GenerationFact` and inserts its +exact generation retention in one Cellule command. Both catalog and refs must be +present. The request needs current Read access, including anonymous public reads; +it does not grant Write/Admin or create a preparation/artifact namespace. The +pin binds repository, logical reader ID, actual owner incarnation/epoch and +admission sequence. Reusing an existing reader ID is a conflict. Replaying the +same original SDK command returns its original receipt, not a new lease. + +The hard cap is 4,096 retained serving pins per repository. Count probes are +bounded, generation lookups are indexed, and the serving table has a generation +index. SQL triggers reject identity changes, replacement, backwards lease +updates and insertion above the cap. `RenewServingPin` requires current Read, +actual owner and an unexpired exact pin. Renewal cannot shorten its deadline or +change its selected generation, even if the current head has advanced. + +Lease expiry prevents serving and renewal. It never deletes retention: an old +worker can remain suspended after its lease or owner expires. Existing generation +reaping excludes exactly the serving-pinned generations and retains the existing +preparation floor protection independently. This is SQL fact retention, not a +completed remote artifact collector. The final typed GC/backup/restore inventory +must include these roots and all other reader/backup/recovery owners. + +## Exact acquisition and renewal custody + +Production acquisition and renewal run through `ReadyServingCommand` and the +existing publication coordinator. Raw commands 44/45 remain domain receivers +for the custody envelope and test fixtures; production does not register them. +Commands 41/42/43 use codec version 2 with fresh schema and MAC domains. There is +no decoder, default purpose or data migration for the old journal format. + +Reuse the existing custody intent, SDK snapshot/body, authenticated carrier, +ordered predecessor, recorded phase, first-writer retirement and shared bounded +queue. The primary key is `(purpose, operation, step)`, with distinct creating +and serving purposes. Pending/grant indexes, exact loads, stop authentication and +scanner keysets include purpose. The same logical ID can therefore name one +creating request and one serving reader without joining their histories or +coordinator jobs. An SDK request identity remains globally unique in the journal. +Historical serving grants never restart a creating namespace. + +Serving intent registration requires current Read rather than Write. Acquisition +uses the existing BeginRequest identity fields for repository, reader ID, request +digest, service account and requested lease; it allocates no artifact namespace. +Anonymous browsers use a pin owned by an authorized service account and remain +subject to their own fresh Read checks. The exact record is not an anonymous +mutation or an account-authentication shortcut. Renewal keeps the acquisition's +logical account/digest and exact token. The domain write, recorded serving result +and SDK acceptance commit together; late errors or ignored phase writes leave +both the pin mutation and SDK acceptance absent. + +The coordinator retains both original registration and execution commands across +absent/lost replies and panics. It uses the shared foreground class with the +custody reservation of 28 KiB; that body does not fit the 8 KiB maintenance +reservation. Dropping an observer or closing the queue does not free retained +uncertainty. Known results release retained ownership before returning credits. +Fresh physical capability construction remains separate from historical receipt +recovery, including after actual Cell owner restoration. + +`ServingPin::ready_renew` acquires a physical-drain guard before preparing the +original. The ready value, held admission, dispatch and uncertain recovery share +that same guard. Closing the pin waits until a proven unexecuted held command is +discarded or the exact original reaches a known disposition. Cancellation of an +observer cannot release it. The production owner must retain and activate/discard +held tickets and drive uncertain recovery. `ServingOwner` now performs that +ownership and automatic renewal; the resident pool now composes that lifecycle. + +The existing bounded custody scanner also visits serving heads and can retire an +expired unexecuted original. A stop records logical closure and never invents an +execution result or releases an accepted serving pin. Reconstruction APIs select +the serving purpose explicitly; creating staging recovery rejects serving actions +and results. Settled history is still stored in SQL and requires the planned +admitted immutable history frames and exact lookup to bound long-term growth. + +## Capability construction and admitted reads + +### Owned producer and accepted acquisition handoff + +`ServingOwner::start` obtains bounded owner admission before spawning a private +supervisor. The supervisor retains its proposed SDK identity, factory plan, +original prepared command and any held ticket outside the restartable worker. +It activates held originals and resolves uncertainty through their exact +evidence. Registration/execution transport loss, worker panic and caller loss +cannot replace an admitted original with a fresh acquisition or renewal. + +Before publishing a successful acquisition to borrowers, the producer calls +`ReadyServingCommand::retain_acquisition`. Only an original local acquisition +can use that handoff; a restored journal command or a renewal cannot. The probe +loads the acquisition's exact ordinal, authenticates its recorded acceptance and +checks its original admission sequence. It then reserves the existing exclusive +physical owner, verifies the still-retained exact SQL pin and historical +generation, and brackets that observation with actual owner checks. A later +renewal must not hide the acquisition ordinal. No absent or denied command is +executed by this probe, and no receipt DTO becomes a physical capability. + +Accepted acquisition knowledge remains available after lease expiry or Read +revocation. This permits retention and authenticated cleanup rather than fresh +I/O: every serving operation still checks current access, exact pin, actual +owner and a conservative lease deadline. Cleanup may use the administrator's +bounded physical-read slot after read admission closes. A released, rebound or +missing row fails handoff; duplicate physical ownership is rejected. + +`ServingSnapshot` carries a private borrow guard and exposes only the generation +fact and admitted headers, bodies, refs and typed graph pages. Clones share that guard until the last clone +drops. Closing a producer refuses new borrows while existing borrows retain their +generation and continue renewal. Renewal is scheduled at one third of the +conservatively observed remaining lease; this is a scheduling policy, not a +guarantee that an overloaded or fenced owner can renew. Read revocation, expiry +or known renewal denial closes new borrowing. Physical roots remain retained. + +After the last borrow, the owner stops producing renewals, joins actual pin +workers and prepares the authenticated exact release. Uncertain releases keep +their original. Only a settled denial allows a new release proof to be prepared +after administrator access is restored. Neither a denied release nor an +observer timeout counts as drain. The last producer handle initiates closure; +`ServingDrainObserver` joins its real completion without keeping admission open. +The supervisor and physical pin outlive detached observers. Loss of authority +can keep cleanup pending; physical fencing/adoption remains mandatory work. + +Owner, snapshot and physical-I/O admissions each use the configured 2–64 node +limit and half-cap account share, with separate semaphores. Owner admission +covers the producer's entire lifetime, including a built original before queue +admission. Snapshot admission covers waiting and returned borrow lifetimes. +Long-lived snapshots therefore cannot exhaust the separate physical-I/O slots. +Read budgets must be shared once per node; repository-scoped artifact/index +clients must be shared across the resident's generations. Producer tasks use a +private tracker so their drain can be joined explicitly; production +must stop and join them before closing the node tracker or publication budget. + +Thirteen focused lifecycle families pass on macOS/Rust 1.98.0. They cover both +formats, six registrar/execution fault modes, five producer restart points, +borrowed renewal and clone lifetime, deterministic lost-ack/revocation ordering, +denied registration/release, closed read admission, original-ordinal handoff, +independent contexts' physical exclusion, canceled observers and blocked actual +artifact-provider I/O. They qualify this component, not resident pooling, +process fencing/adoption, production reader conversion or large-team capacity. +The owner fixtures use certified initialized empty catalogs and missing-object +lookups. Full nonempty native/body/history reads require their own production +conversion and qualification; suspended real provider I/O proves drain ownership, +not full-repository serving performance. + +`ServingContext` is explicit trusted configuration: real CellClient/target, +actual PreparationAuthority, shared CatalogIndexes/CatalogFiles, shared node +read budget/TaskTracker, and repository administrator identity. Passing decoded +catalog or generation data cannot construct a serving capability. `ServingPin` +opens only after a fresh exact-pin query and actual owner checks. QueryContext +does not expose its target/owner; SQL repository identity scopes the query, +and the service verifies the configured target and fresh actual owner before +artifact I/O. A pin query result alone grants no serving authority. + +`SelectServingGeneration` (query 48, codec 1) observes the current joint +catalog/ref head under current Read access. Input is bounded at 1 KiB and output +at 512 bytes; repository identity scopes the query and its head lookup uses the +singleton/catalog primary keys. It returns no root before joint initialization. +Selection allocates no pin or creating namespace. Its `GenerationFact` is only +an observation: callers must acquire/check an exact serving pin and verify the +actual owner before artifact I/O. A head advance never changes an existing pin. + +Every exact pin also has one process-wide physical owner. A private Arc guard +is reserved before tracked construction and retained by the pin, detached +workers and release proofs. Duplicate construction is refused even through +independent contexts/budgets. Only cloning that same capability shares its drain +counter. Weak entries are pruned under a short synchronous mutex; the process +hard cap is 4,096 live owners, independent of the per-repository SQL cap. No +provider/Cell await runs under that mutex. Guard loss permits reconstruction +only after every previous local owner/worker/proof has dropped; SQL and actual +owner must then be rechecked. This local exclusion registry is not a durable +acquisition ledger or a substitute for process fencing and restoration. + +The serving budget is explicit, 2–64 concurrent workers, with the existing +nonwaiting node/account admission and half-cap account share. Production must +create it once for the node and pass clones, not create a new budget per request. +Closing prevents new work; it does not abandon an accepted read. Production must +stop and join admission producers before closing and waiting its TaskTracker. + +The implemented read operation returns at most 512 authenticated object headers. +It validates OIDs, obtains admission, increments an owned physical-drain guard, +and spawns through the supplied TaskTracker before yielding. The tracked worker +reobserves current Read/exact pin/fresh owner, opens the actual certified catalog, +performs the existing bounded metadata batch and rechecks authorization and lease +before returning. A conservative local deadline starts before the lease query; +query/provider delays cannot extend the lease. An expired or revoked result is +refused after owned I/O finishes. Cancellation detaches the observer; it cannot +drop the tracked worker, admission or drain guard. There is no timeout that drops +an owned metadata/SQLite future. + +The pin retains one lazy reader; configured index/file clients share their bounded +caches across generations. This does not expose raw catalog readers or native +workspace mutation authority. Object bodies, native operations, response streams +and all current object/ref/cache/graph/browser consumers still need conversion. + +## Bounded resident generation pool + +`ServingPool` retains at most four generation slots, including acquisitions and +closing owners. Pending viewers coalesce on one acquisition; known owners are +matched by their accepted token's generation. Current-root selection is an +admitted, tracked observation under the requesting viewer's Read access. A head +advance between selection and acquisition returns the actual accepted joint +fact, never a capability labeled with the earlier observation. Product consumers +must resolve their revision/ref through that accepted snapshot. + +The pool admits each viewer before spawning private request work. Its permit +covers selection, acquisition waiting and the returned snapshot borrow. Observer +cancellation detaches accepted work, and the owner's original stays retained. +At capacity, the pool initiates closure of one least-recently-used unborrowed +owner and returns an explicit capacity error. Its slot remains charged until the +real producer has exited; there is no unbounded retired-owner list or waiting +behind old provider work under the pool lock. Borrowed old generations remain +immutable. Their independent owners keep renewing while other generations work. + +Eviction pauses acquisition/borrowing and uses a nonwaiting owner handshake. +The producer driver must be idle, with no pending original, outstanding borrow +or physical pin work. An already built/held/uncertain command is never discarded +by that handshake. Busy refusal resumes the same owners. Only after all owners +are paused may the common coordinator reserve the bounded exact-token drain +gate. Accepted drain is owned by a private task: cancellation cannot abandon +its guard or strand paused owners. Actual releases and producer joins precede +coordinator closure. Repeating quiescence after a later Cell-release refusal +works only after every owner/request has really drained. + +`RepositoryManager` owns one 64-slot node serving budget. Each initialized local +resident's `RecoveryServices` owns one pool/context and repository-scoped shared +index/file clients, using the actual NodePeer authority and its existing common +publication coordinator. `RepositoryCell` holds a weak pool association; stale +repository handles do not keep serving admission open. Remote routes receive no +local pool. Partial service construction explicitly joins its pool/scanner before +returning failure. Last pool-handle loss initiates its private supervised drain. + +Service registration and shutdown inventory share the manager's loaded-resident +mutex. Shutdown cancels the permanent construction barrier under that mutex +before collecting pools. A constructor registers its lifecycle owner before +exposing the weak repository capability, under the same mutex. A constructor +that loses this race joins its private pool, scanners and exact recovery inside +its already tracked residency task, then returns `CellDraining`. No unpublished +pool can become accessible or escape the shutdown inventory and task join. + +Recovery quiescence pauses discovery before pool drain. Server shutdown joins +HTTP/SSH ingress, closes and joins serving pools while Cell, heartbeat and +publication admission remain available, then closes the node task tracker and +drains recovery/Cell/workspace. Node serving admission closes only after pool +drain. The standalone recovery drain also enforces that ordering. A borrowed +snapshot clone or detached real I/O cannot permit early publication-budget +closure, Cell shutdown or workspace reuse. + +Six pool and three real-manager families pass as part of 56 focused serving/ +resident tests. They cover concurrent viewer coalescing and current access, +canceled cold observation/lost acknowledgement, four borrowed generations and +actual slot reuse, deterministic acquisition head races, busy/canceled exclusive +drain, blocked old provider work with independent other-generation release, weak +repository access and real shutdown retaining publication/Cell/heartbeat/workspace +until the last borrow. These empty/copy-root fixtures qualify ownership and +selection semantics, not native publication throughput or full-history serving. +The third manager case deterministically pauses a constructor before publication, +proves that its public capability is unavailable, starts actual server shutdown, +and verifies that workspace/publication ownership remains until rejected +construction cleanup joins. The regression fails against the preceding weak +association ordering and passes with the registration barrier in both formats. + +## Certified immutable ref reads + +`ServingSnapshot::resolve_ref` and `refs_page` use only the ref snapshot in the +accepted joint fact. The snapshot descriptor is loaded once per retained pin, +validated against repository, object format and the joint generation, and shared +by its viewers. The resident's existing `CatalogIndexes` also owns the shared +`RefStateIndex`, so unchanged authenticated nodes reuse the same bounded cache. +Public descriptors or a cache hit do not grant Read or retain a generation. + +Ref and canonical-header reads share a private admitted worker. Current access, +owner and lease are checked before and after artifact work; expiry or revocation +refuses the result. Observer cancellation detaches that worker without returning +its admission or physical guard. Closing a pool cannot release its pin while a +ref descriptor/index download is still pending. The worker finishes its actual +I/O rather than dropping it to manufacture a timeout/drain result. + +The new path reuses `RefPage` and `RefExpectation`. Lookup distinguishes a name +that never existed from a retained deletion with a version. Page size is at most +256 records and 512 KiB of name/record charges. Live cursors skip authenticated +zero-weight subtrees; other consumers can request deletion versions. Continuation +requires the first page's ref generation, and a different selected ref generation +returns an explicit changed result. An old borrowed snapshot continues to read +its old immutable refs after the current head advances. Ref generation is the +snapshot's counter, not the catalog counter; catalog-only changes can share refs. + +The local browser's `Refs` and `Resolve` operations select this service. Both the +default branch and its tip come from the same immutable snapshot; legacy `refs` +and `ref_generation` rows cannot override them. The HTTP request ceiling is +512 KiB so even supported long ref cursors with JSON escapes can be submitted. +Names above the index's 65,535-byte limit reject as invalid input. Changed page +generations return HTTP 409. This does not yet convert remote owner routing, +object bodies, tree/file/history browsing, native Git or policy/default-branch +producers; those remain required for the full cutover. + +Five new ref families and one actual HTTP/manager family exercise both object +formats: count/byte continuation, tombstones, old-generation immutability, cached +and in-flight revocation, canceled real provider I/O, malformed snapshot context, +and immutable authority despite deliberately conflicting legacy SQL. The ref +inventory fixtures inject roots to isolate reader behavior; their tips do not +qualify native graph publication or capacity. Final frozen-source totals are +recorded in the [implementation status](../large-repository-implementation-status.md). + +## Sticky closure and exact release + +Closure prevents new reads and waits for every owned drain guard. Notify +registration precedes active-count observation, avoiding a lost final wakeup. +Only after physical drain may the private owner mint a purpose-separated MAC +proof binding tenant/application, exact pin and administrator. The release +receiver checks the MAC, current Admin, actual owner and exact row. Lease expiry +does not prevent a drained release. + +Release uses the existing publication coordinator's reserved maintenance share, +with an 8 KiB command reservation and 1 KiB input/128-byte output bounds. A +separate job kind preserves the actual reader ID without colliding with a +creating publication or custody retirement that has the same logical ID. The +owner caches an Arc of the original prepared SDK command. Repeated factories +reuse that original even if the caller supplies a different proposed identity. +Absent/lost-reply/panic recovery resolves its exact evidence and retains credits +through uncertainty. Known outcomes clear the retained command before credit +return; a known successful release permanently closes the pin. A caller cannot +reopen it by replaying acquisition or by cloning an old receipt. + +An eviction owner can reserve `ServingDrainAdmission` only while the common +coordinator is idle, after pausing its serving producers and borrows. The +reservation accepts at most 16 distinct exact tokens from that repository and +admits only their privately prepared releases. Matching a reader ID alone is +insufficient: owner, original admission sequence and generation must also match. +Current receiver authorization and physical-drain proofs remain mandatory. +Busy reservation leaves all existing admission and commands unchanged. + +The guard closes the coordinator only after every selected token has an observed +successful release and no held, dispatched or uncertain work remains. Denial, +absence and detached observers cannot satisfy this condition. Global queue +closure waits for the guard to finish or be dropped so selected releases can +still be admitted. Guard cancellation resumes ordinary admission but never +cancels accepted work, returns its credits or reopens an already closed queue. +The caller must keep serving producers paused through guard completion/drop. +The resident pool wires this scheduling primitive into production eviction; +production shutdown keeps the node publication budget open until releases finish. + +A new owner cannot renew/release old-owner pins merely because its epoch is newer. +They remain roots until actual physical fencing/drain and an authenticated +adoption/release protocol is implemented. Conservatively retaining abandoned +roots preserves correctness but does not establish operational quota recovery. +Exact acquisition/renewal command reconstruction is implemented below. Automatic +production handoff, physical fencing/adoption and abandoned-root quota recovery +remain required. Do not compensate with automatic expiry deletion or a synthetic +owner fence. + +## Production integration and qualification gates + +The resident pool now integrates `ServingOwner` and its accepted acquisition +handoff with manager residency. Command reconstruction alone does not establish +physical ownership. Native bodies and local browser/comparison reads now carry that ownership. +Remaining transfer producers must carry it through complete response streams. Actual +eviction and shutdown already own pool drain. A close must join all producers and +workers before Cell/workspace/artifact release. + +Consumer conversion must preserve these boundaries: + +1. `RepositoryManager` owns the node read budget. Each local resident owns one + repository-scoped context and bounded generation pool, sharing its index/file + caches. Coalesce acquisitions rather than constructing a producer per viewer. + Query 48 observes a head; acquisition selects and retains its actual accepted + fact atomically. A head race must not associate a producer with an earlier + observation's generation. Product reads bind to the accepted joint fact. +2. Eviction pauses acquisition/renewal producers and new borrows before reserving + exact drain admission. A pause handshake must account for built/held/uncertain + originals; merely toggling a boolean cannot establish an idle coordinator. + Refusal resumes the same owners. One blocked old-generation worker must not + serialize unrelated live-generation renewal. +3. Shutdown stops borrowing and joins all generation producers while Cell, + administrator authority and publication admission remain usable. Only then + may the node tracker/publication budget close and resident recovery/Cell/ + workspace drain finish. The current shutdown path enforces this ordering; + remaining native/stream consumers still need to carry the snapshot guard. +4. Actual process fencing and restored-owner adoption must precede releasing an + abandoned pin. A historical lease or an expired deadline is insufficient. + Quota recovery must use that authenticated lifecycle rather than reaping SQL + roots based on expiry. + +Regression families exercise Read/public access, joint initialization, original +acquisition replay after release, token scope, revocation, expiry, monotone +renewal, generation reaping, schema quota/identity guards, blocked real provider +I/O, observer cancellation, actual owner restoration, bounded codecs/MAC domains, +and exact release absence/lost acknowledgement/panic. Registered acquisition and +renewal families exercise all six registrar/execution transport fault modes, +held/canceled observation, closed recovery, late/ignored atomic rollback, +recorded revocation, cold owner restoration after SDK expiry, both-purpose +page-one scanning, shared pending quota and v2-only bounded codecs. Initialization retention is +retired through its actual registered terminal release so an unrelated floor +cannot conceal a serving-retention bug. Trusted generation/quota SQL fixtures +qualify receiver invariants, not native publication or team capacity. + +The current full workspace library still has five failing unconverted `objects` +readers. Production producer/reader/final DDL conversion, admitted immutable +custody history and exact lookup, physical input takeover, scanner restart/fault +campaigns, typed GC/backup/isolated restore, OS CPU/RSS/I/O/PID containment, native +acceleration/physical rewrite/fair maintenance, signed native completion/cold clone, +[file attribution](file-attribution.md), and full Linux/Kubernetes/Chromium plus +10,000-SDE mixed-load/recovery/capacity qualification remain mandatory. + +## Certified bounded native body reads + +`ServingSnapshot::body(oid, limit)` resolves the object through its accepted +catalog and preferred source before opening native Git. A missing catalog entry +returns absent even when an old cached pack physically contains that OID. The +foreground copy limit is at most 64 MiB; larger objects require a separate owned +streaming producer rather than increasing this allocation without bounds. + +Each resident's configured `CatalogFiles` shares immutable native pack copies +across borrowed generations. Sixteen fixed load stripes coalesce equal misses; +at most four completed copies are cached and eight file slots are live, +including borrowed/evicted copies and deferred cleanup. Idle copies can be +removed to make room under the shared disk budget. Fully authenticated pack and +index bytes are reserved before download, and their Git checksum/index binding +is checked before they enter the cache. Whole-pack verification occurs on a +bounded admitted blocking job, once per retained copy. + +A native body must match the selected canonical kind, size, Git object ID and +BLAKE3 body digest. Header mismatches and limits reject before allocation. The +batch is poisoned across an incomplete, canceled or invalid read and becomes +reusable only after a complete verified frame. Native process ownership includes +its physical-generation guard and cache reference through descendants and +leader reaping. Blocking creation, writes and hashing also retain their work +owners independently of observers. A cache's file-slot/root owner follows failed +or deferred cleanup; its generation guard is not retained by the idle cache. + +Resident configuration uses the node's shared foreground native resource scope. +A metadata-only loader has no implicit native resource pool and refuses body +reads. Native cache statistics expose live/cached files, cache hits and completed +downloads. Authorization and conservative lease checks still run before and +after work through the common serving read contract. + +This API is a prerequisite for browser, graph and transfer conversion. Local +browser body/commit consumers now use it; remote routes still require conversion; it is +not evidence of completed native streaming, OS containment, publication or +large-repository capacity. Pack reuse reduces repeated downloads, but each +bounded body currently starts a native batch process. Shared persistent readers +and multi-object batching require separate bounded scheduling and qualification. + +A private source pack can omit graph parents that reside in other certified +sources. It is sufficient for verified object-body decoding, but is not by itself +a complete native history workspace for `git log` or attribution. Such producers +must hydrate the required graph through certified source selection and retain +all physical inputs through their owned lifetime. Cold load still reads and +verifies a full pack/index pair; these tests establish reuse and bounded ownership, +not a cold-read latency guarantee for multi-gigabyte packs. + + +## Certified browser objects and typed ancestry + +A local browser tree, file or first-parent history request borrows one accepted +joint generation for its whole view. Annotated tags, commit/tree parsing, sizes +and file previews read only certified headers and verified native bodies through +that snapshot. The selected catalog determines presence even when a reused pack +contains more objects. The public 32-entry pages, literal raw-byte paths, mode +handling and 256 KiB preview bound remain. Wrong-format inputs reject as invalid; +zero or absent commit IDs return missing rather than a storage failure. + +Comparison files, previews and patches also borrow one generation for their +object and ancestry reads. `ServingSnapshot::edges_page` accepts one to 128 +strictly sorted unique nonzero IDs in the repository format. It returns at most +512 typed edges and headers for the parents visited, including explicit absent +headers. A `(parent, child)` cursor resumes within a parent and then advances to +later requested IDs; it must identify a parent in the same requested set. +Continuation headers may repeat the cursor parent. An exact-full page returns a +conservative continuation and can require one final empty read. These pages are +not ordered Git parent lists; commit bodies retain ordered parents for history. + +Edge source selection uses the accepted preferred metadata, never legacy Cell +object/parent tables. Local immutable SQLite edge queries run off the executor +with a child physical guard. The common admitted read checks access, actual owner +and conservative deadline before and after work. Observer cancellation detaches +the tracked worker; generation release still waits for its real provider/query +completion. Edges and headers use the same shared authenticated file/index cache. + +Merge-base traversal expands groups of at most 128 certified commits, consumes +all edge continuations, and excludes tree edges by expected kind. It rejects +missing/non-commit roots, including equal tips, and retains the existing bounds +of 100,000 commits and 250,000 parent edges. Best common ancestors are computed on +owned bounded graph data. It starts no native process per traversed commit. +Current pull/review/thread metadata authorization still reads its existing Cell +records and live legacy ref rows; their producer/authority conversion is open. +This checkpoint does not establish a fully converted pull lifecycle. + +Production HTTP tests use native packs physically verified into metadata shards, +then trusted installation of the joint catalog fact. They exercise both object +formats, raw paths, directory/history continuation, modes, tags, file previews, +merge comparisons and a 532-parent native merge whose relevant parent is beyond +the first edge page. Suspended-provider and revocation tests qualify edge-worker +ownership and cached authorization. This isolates consumer behavior and is not +end-to-end live producer publication, cold latency or large-team qualification. + +## Complete native read workspaces + +`ServingSnapshot::workspace` takes a bounded sorted set of certified object roots; +its caller must independently authorize those roots. `ref_workspace` instead +chooses every live ref directly from the retained joint ref index. It streams +32-name pages into admitted native `packed-refs` and the graph frontier, with no +repository-sized Rust ref map or root-count cap. Empty live refs yield an empty +native repository. HEAD uses the same immutable ref snapshot's default branch. + +Construction resolves preferred object sources through the certified catalog. +It downloads each distinct, authenticated native pack/index pair once directly +into one unpublished cache and verifies its physical binding. It never decodes +all blobs into loose files. A disposable admitted SQLite spool tracks frontier, +expected types, completed membership and input deduplication. Frontier batches +are at most 128 objects; typed edges and membership lookups are bounded to 512. +Gitlinks do not become graph dependencies. A physical pack may include unrelated +objects, so `contains` and bounded verified bodies use completed graph membership, +not native pack presence. Local HTTP/SSH wants use this membership before Git +receives them. Filters remain validated before native traversal. + +The spool reuses admitted SQLite growth: reserve database, journal and overhead +before allowing additional pages, replay only rolled-back transactions, and +refuse growth without retaining a successful prefix. Its maximum and page cache +are explicit. Input deduplication requires identical pack/index digests, sizes +and object count for a shared native checksum; a namespace change alone does not +require a second physical copy. + +Long construction retains a producer borrow and exact pin. Bounded steps refresh +lease/owner/access observations at most every 250 ms, and always reobserve before +returning. Valid renewals can carry construction beyond its initial conservative +read deadline; an expired exact pin cannot be resurrected. Ordinary short reads +retain their existing original deadline. Cancellation detaches the observer and +keeps provider jobs owned until real completion. + +Construction read admission covers the worker and its blocking/provider jobs. +Completed workspaces release that read credit while retaining snapshot ownership, +physical guards, native file admission and disk reservations. Subsequent reads +acquire fresh read admission. Native subprocess owners retain the workspace +through descendants and physical cleanup, including deferred/quarantined cache +removal. Lease expiry alone is never physical drain. + +This is a correct bounded construction path, not a demonstrated hot repository +cache or large-team capacity result. Each transfer currently constructs its own +workspace and downloads selected inputs. Sharing/coalescing authorized generation +workspaces, cold I/O acceleration, fair scheduling and history-sized workload +qualification remain required. Legacy generated/write producers still have +object-table hydration and publication paths to replace. diff --git a/docs/design/durable-custody-command-intents.md b/docs/design/durable-custody-command-intents.md new file mode 100644 index 00000000..f0a2f201 --- /dev/null +++ b/docs/design/durable-custody-command-intents.md @@ -0,0 +1,101 @@ +# Durable custody command intents + +Status: local, unpublished production cutover. Repository initialization and the staging/preparation service factories use this protocol. Cold service owner fencing/reconstruction, compact terminal archival and capacity qualification remain required before release. + +A prepared command can be lost before Begin grants an artifact namespace. A final publication's existing [registered recovery root](mandatory-publication-registration.md) cannot cover that interval: its body artifacts require an independently admitted namespace. Allocating a fake namespace or reconstructing a fresh SDK identity would cross the custody or exact-command boundary. + +## Representation and bounds + +The protocol reuses `PreparedCommandSnapshot`, `Stamp`, `Recorded`, `CertificateEnvelope`, the existing Begin/Claim/Renew/Bind domain logic, namespace allocator and independent generation pins. One additional `WITHOUT ROWID` relation, `catalog_custody_commands`, stores command metadata before and after admission. It stores no Git object, object edge, physical pack or inventory entry. Its rows are proportional to custody transitions. + +Each row is keyed by logical operation and ordinal. A separate unique key binds the original incarnation and SDK request ID. The authenticated intent binds tenant/application, repository, actor, logical operation/digest, original SDK stamp, ordinal, exact predecessor digest and the snapshot/body digest. The typed body distinguishes preparation Begin/Claim/Renew and staging Begin/Claim/Renew/Bind. It preserves the original encoded bytes, compiled contract, incarnation, identity and expiration. Neither restoration nor registration extends that expiration. + +The input is bounded to 1 KiB, SDK snapshot to its existing 2 KiB ceiling, authenticated carrier to 1 KiB, complete stored intent to 4 KiB and recorded phase to 1 KiB with a 512-byte typed reply. Every factory and receiver applies its complete encoded limit; individual ceilings do not authorize a sum exceeding the complete limit. Factory encoding fails before registration or allocation if the combined representation exceeds it. Actor identifiers retain their existing 64-byte limit. Ordinals use the existing recovery protocol's 65,535 ceiling; exhaustion refuses preparation rather than replacing history. + +At most one unresolved, unretired row exists per logical operation. Indexed admission counts at most 1,024 pending heads; settled history does not consume this unresolved-work quota. A partial index serves that count. The primary key serves latest-head and exact-ordinal discovery. An explicit indexed lookup on operation/incarnation/admission sequence discovers an authentic historical grant for restart Claim without scanning the operation's renewal history. SQL guards prevent changing or replacing an intent, changing a settled phase or its grant identity, and deleting retained knowledge. + +## Registration and execution + +Command 41 registers the authenticated exact intent. The first matching ordinal wins. Advancing requires a settled or explicitly retired predecessor, its exact encoded digest and matching logical actor/digest. Unknown work cannot be skipped. A new registration checks current Write, the actual incarnation and original command expiry. Exact existing knowledge remains discoverable after permission or owner loss; this grants no execution or upload permission. + +The factory retains its original snapshot/body through registration. A private query of the exact row proves registration even after a lost acknowledgement or registrar SDK expiry. It returns an identical winner before issuing another registration mutation. A missing or corrupt row after uncertainty remains an error. A competing candidate cannot dispatch its original command. New registration knowledge is observed through the authoritative Cell query, and execution independently verifies the same pointer. + +Command 42 accepts only the registered original SDK stamp and typed body at that ordinal. It reuses the domain receiver's current authorization, actual owner fence, exact token/pin, expiration, generation and quota checks. Namespace/pin/domain writes and the original typed result share the same transaction and SDK acceptance. SQL or encoding failure rolls back all of them. Positive results and domain denials both commit an authenticated-intent-bound phase; private service boundaries normalize trusted negative replies back to `Rejected` while preserving their original receipt. + +Recovery queries the original ordinal and observes its phase before SDK resolution or any current-custody query. Known history returns the original receipt after later renewals, Claim, owner loss, revoked Write or SDK expiry. A phase missing despite SDK acceptance is treated as uncertainty. Only authoritative SDK `Absent` can restore and execute the original unchanged bytes. `Unknown`, `Expired`, query failure and corrupt metadata retain the original evidence. In particular, an expired unresolved command is not replaced with a new identity. + +Historical grants are knowledge, not leases or artifact retention roots. They never restart a clock or retain every old base forever. Fresh authorized queries and actual independent pins determine current custody. After a grant's operation and pin are reaped, Claim can authenticate that exact indexed historical token, recheck current Write/logical availability/quotas and allocate a different namespace/pin under the actual executing fence and sequence. It cannot displace an active successor or recreate a completed outcome. + +## Service ownership + +`OwnedCustody` retains the exact original 42 and original registrar 41. The StagingCoordinator's Begin/Claim/Renew/Bind and bound Claim/Renew variants use that owner, as do ReadyPreparation factories in the fair PublicationCoordinator. An ordinary lost registrar reply can be resolved by the identical authoritative pointer. Registrar uncertainty and worker panic retain original execution evidence and both command bodies; observer cancellation or closing cannot replace them. Cold restore loads a registered head without issuing a new registrar or original identity. + +Each admitted custody job reserves 28 KiB, including retained/dispatch intent bodies, registrar transport/query decode and original body/reply ceilings. The staging checkpoint slot adds its existing 4 KiB, totaling 32 KiB. This accounting leaves operation, actor, worker, class and global byte caps unchanged. It does not measure the entire process heap or native resource use. + +Known phase results precede SDK resolution and local custody guards. Only proven absence can invoke an original under a still-valid local fence/deadline/ceiling; expired or unknown evidence stays retained. Fresh post-result queries establish a conservative clock, never from recorded reply timestamps. Direct caller-owned raw session/base renewal methods have been removed; ready renewals transfer into the service-owned coordinator. + +Session construction now requires a mandatory server-owned `PreparationAuthority` bound to the exact repository Cell target. Production reuses `NodePeer`'s validated durable Cell Control and live node advertisement; a decoded lease or supplied epoch cannot construct this capability. Owner incarnation and epoch are checked before and after the lease query, and its conservative monotonic deadline starts before those observations. Staging probes, bound handoff, preparation Claim/Renew restoration, base catalog loading/frontier selection and standalone positive root recovery retain the same authority source. No optional legacy source exists. Local runtime qualification fixtures read actual durable Control records, with their network-advertisement difference explicit and absent from production builds. + +Fresh observations add durable Control/live-advertisement reads at custody and catalog-selection boundaries, rather than per Git object. Their existing bounded readers limit persisted inputs, but those I/O buffers are outside the 28 KiB command-wire reservation. Count their resident memory, I/O and tail latency in mixed-load qualification; an unvalidated owner cache cannot replace them. + +A missing, corrupt, unowned or changed authority observation prevents fresh custody. A failed refresh or base observation permanently fences the existing shared session; repairing the durable source cannot un-fence it. Original known outcomes still resolve first and remain recoverable even when no usable session can be returned. Explicit registered Claim under the current owner can allocate and restore a new usable session. An owner observation is not a lease on ownership or an atomic publication check: final receivers retain their actual-owner transaction checks. Cold staging reconstruction and shared-fence callback drain now have focused qualification. Bounded expired-head retirement is described below; production takeover wiring, retained-input adoption and full resource/scale qualification remain release work. + +## Separate retirement of expired originals + +Command 43 (`StopCustodyIntent`, 1 KiB input / 128-byte reply) closes an expired unresolved original without claiming that command 42 executed. Its private factory seals the exact repository-cell tenant/application, operation/ordinal, original intent digest and freshly observed actual owner in the existing `CertificateEnvelope`. The receiver authenticates this purpose-specific carrier, reloads the exact original, and checks expiry using receiver time and the actual executing owner. Current ownership is server authority for metadata retirement; it does not depend on the original actor retaining Write. A live original is refused. An original that committed first remains settled, and an earlier stop remains immutable. + +A separate, authenticated `stopped` record binds the original intent and contains the stopping owner's fence, timestamp, stop command stamp and the shared `Recorded` result/sequence. It is at most 1 KiB, cannot coexist with an execution phase or grant identity, and cannot be replaced or cleared. This record removes only pending-head quota. It neither deletes the original bytes nor releases an artifact namespace, independent generation pin or remote object. SDK expiry remains expiry; original execution recovery remains pending when no original result exists. `RegisteredCustody::closed` and `stop_fact` expose logical closure separately from `settled`. Frozen original command 42 cannot execute after a stop, and `OwnedCustody` reports the typed stop fact so exact staging recovery can fence/drain and return local admission without inventing an original reply. + +`CustodyStopOutcome` retains the original custody evidence and the separate retirement invocation evidence. A matching recorded stamp recovers that retirement invocation's original receipt before SDK expiry or current authorization checks. If a competing stop won, it returns that first-writer stop fact with `committed=None`; it never attaches the winner's receipt to an unexecuted or unresolved losing SDK identity. Exact absent/lost/panicked retirement invocations stay owned by the existing dispatcher through observer cancellation and closure. Marker-query failure cannot authorize dispatch. A later explicit custody successor may advance from closed history while preserving the prior original's identity and lack of result; the domain receiver still decides whether its requested Begin/Claim/Renew is valid. + +`CustodySupervisor` reuses `RecoveryScanLimits`: pages contain at most 128 operation keys, with a bounded 10 ms–60 s interval. The partial `catalog_custody_pending` index seeks byte-ordered keys where both phase and stop are absent. Each original is authenticated separately. Bad heads consume a bounded failure observation and advance the cursor, so later heads remain visitable; wrapping revisits bad heads and newly registered keys behind the cursor. Only expired unresolved heads are eligible. The existing account/class-fair `PublicationCoordinator` owns each ready stop under its reserved maintenance slots and 8 KiB command-wire reservation. A separate dispatcher key kind for retirement preserves the real operation ID and allows a stop alongside its own pending preparation. Duplicate stop work defers admission. Existing operation/account/class byte and worker caps still bound both jobs. Known closure requeues only a preparation whose exact original evidence matches, and that preparation's shared session is fenced before admission returns. Recovery sweeps only uncertain stop jobs in that bounded queue, including accepted stops whose keys have disappeared from the SQL scan. No new unbounded local outbox or fabricated artifact namespace is introduced. + +Dropping or shutting down discovery stops between visits and joins its current scan, without canceling already admitted retirement commands. Close/drain the dispatcher separately; unresolved commands retain their original evidence and reservation. General production repository scanner lifecycle wiring, automatic resumption of stopped staging consumers and full mixed-load qualification remain required. The production initializer now handles its one admitted head as described below. After process loss, an existing marker proves logical retirement. Without a marker, a new idempotent retirement request may compete for first-writer closure, but cannot rewrite the original custody command or turn its unknown outcome into a result. Durable exact recovery of the retirement transport itself is distinct from that logical first-writer fact. + +Eleven regression families cover both object formats and all seven custody kinds; real owner restore after local SQLite removal and retirement SDK expiry; immutable first-writer records; live/forged/stale-owner refusal and accepted-original races; ignored/aborted late-write rollback; absent/lost/panicked/private-query-failed dispatcher recovery through closure; reclaiming exactly one pending slot at the 1,024-head limit; bounded indexed scans past 300 corrupt heads and revisiting earlier keys; accepted-stop recovery after scan-key removal; existing uncertain staging recovery; rejection of an authenticated marker transplanted to a different original; and stopping an expired renewal held in the same dispatcher while fencing its still-live shared session. These are small protocol/capacity-boundary fixtures, not repository or team throughput results. + +## Startup integration + +Pending repository initialization discovers its latest registered custody head before constructing another original command. It recovers pre-dispatch Begin, accepted/denied Begin and subsequent Claim/Renew results. A matching current owner and fresh exact custody query are required before using a historical grant. Otherwise, a known resolved phase can precede an explicit registered Claim. A known denied final initialization forces Claim of that refused attempt even when its old Begin grant is still readable; a previously accepted successor Claim is recovered rather than repeated. Unknown phases stop initialization. + +The certified final initializer, its immutable root graph and [terminal retirement](terminal-publication-retention.md) remain the authority before repository Ready. The existing tracked repository transition owns startup through cancellation. Ready restore observes the immutable initialization and preserves the original intent/phase bytes without allocating another namespace. There is no new product API or permission granted by these private metadata queries. + +## Production initialization after logical retirement + +The production repository transition now discovers its one deterministic initialization head before new custody registration. It validates initialization purpose and logical context. A settled phase remains execution knowledge, including after SDK expiry. An expired unresolved original is first recovered exactly; only the same original's unresolved evidence allows preparing retirement. The authenticated stop receiver independently checks actual ownership and its clock. Missing/corrupt current-owner authority refuses retirement and retains the original. Unexpired absent originals continue through their existing exact execution path. + +Startup reuses the same retirement factory, sealed input, first-writer record and exact-result logic as the maintenance dispatcher. Its crate-private `complete_tracked` entry point runs inside the existing account-bounded repository transition, which the manager's task tracker owns through HTTP observer cancellation. It visits only that initialization head and does not create a background scanner, dispatcher job or outbox. The maintenance dispatcher's 8 KiB reservation describes its own service path, not this transition's total resident memory. Native preparation cannot begin until closure is observed and a separate newly registered custody receiver grants fresh authority. + +After authenticated closure, startup reads the current operation at or after the stop's receipt. The indexed singleton/operation lookup validates repository ID, object format, owner, actor, digest, preparation generation and token fields. A wholly absent operation permits Begin; a matching current initialization attempt selects Claim. Mismatched or malformed state is an error rather than absence. The original stopped command keeps its bytes, identity and unresolved SDK result; startup never fabricates an original denial or receipt. The successor is explicit journal history and still undergoes current Write, actual owner, token/pin, namespace and quota checks. A newer failure is handled using that successor's action, not attributed to the old stopped action. + +Registered final initialization is resolved before this path. Unknown final publication cannot be skipped using an unrelated custody stop. Ready repositories still require their retained certified initialization and never create an empty catalog in response to missing metadata. The local staging service now observes exact stopped ordinals automatically through its [bounded retirement probe](staging-service-lifecycle.md#automatic-observation-of-custody-retirement). General resident-repository scanner ownership/drain, retained-input adoption and settled-history archival remain required. + +Seven regression families exercise both object formats through the production registry/schema and actual initializer. They cover manual closure and automatic retirement, preserving unknown original results, live absence and known grants after SDK expiry, stopped Claim/Renew with genuine owner restoration after deleting local SQLite (including retirement under the restored owner), wrong purpose, inconsistent actor/digest/identity, and missing/corrupt durable Control. The final focused run passes in 16.12 seconds. The full frozen-source workspace library run passes 592 cases and fails only the same five unconverted legacy-reader cases; all 307 publication and seven startup cases pass within that run. Nine real startup/workspace lifecycle cases additionally pass in 3.48 seconds. Workspace/all-target Clippy with warnings denied, the server build, formatting and static protection checks pass. These results qualify this startup increment, not the full producer/reader cutover or large-team capacity. Evidence is `/tmp/canopy-startup-stop-validation.json`. + +## Automatic local staging closure qualification + +The existing staging coordinator now observes authenticated stops for each exact admitted original through one read-only bounded probe. It reuses indexed exact lookup and the existing recovery/fence/drain path, preserves unknown command-42 evidence after expiry, and excludes checkpoint/final owners. See the [staging lifecycle contract](staging-service-lifecycle.md#automatic-observation-of-custody-retirement). Seven added regression families and the previously failing automatic-closure regression pass. Final frozen-source qualification passes all 314 publication and seven startup cases within the full workspace library; that broader suite still fails only the five known unconverted legacy readers (599 pass, five fail). Nine real workspace/lifecycle cases, warnings-denied Clippy, the server build, formatting and static protection checks pass. These are local protocol/lifecycle results, not full-history or team-capacity proof. Evidence is `/tmp/canopy-stage-stop-validation.json`. + +## Cost and release work + +Each new custody transition currently adds one registration mutation plus one execution mutation. Exact known lookup/replay adds no execution mutation; already registered retries avoid another registration mutation. Count these phases, policy/native checkpoints, final registration and completion in serialized service-time and fairness budgets. The earlier two-command illustration is not this protocol's total push cost. + +Inline command metadata closes the pre-namespace correctness gap, but retaining one SQL row per renewal forever is not the intended final storage strategy. Before release, compact settled per-operation history into bounded immutable frames in a genuinely admitted namespace, reusing the existing saved-command/root/frame codecs and indexed immutable storage. Keep an authenticated discoverable SQL head and retain exact historical lookup; denied pre-admission work cannot depend on a fabricated namespace. Include this history in typed collection, backup and isolated restore, without treating the historical grant's base descriptors as new live roots. The current implementation conservatively retains rows and does not claim repository/team capacity. + +The staging and preparation factories now retain this protocol through admission, cancellation and service closure. Cold registered-head reconstruction and shared-fence callback drain have focused qualification; complete production reconstruction/adoption and OS resource ownership before release. Remove raw custody bindings from production; domain methods remain callable inside the registered receiver and explicit qualification fixtures only. Remove redundant first-admission columns after their consumers and restart proofs use this journal. Complete foreground producers/readers, final schema removal, serving-generation retention, typed collection/backup, OS resource containment, continuous maintenance, physical rewriting and full-history mixed load before publishing the hard cutover. + +Qualification covers SHA-1/SHA-256 first-writer races, pre-namespace persistence/discovery, unregistered and losing identities, late registration/phase rollback with SDK absence and exact retry, immutable metadata, all seven transitions, historical receipts after successors, original denied Begin/Renew after real SDK expiry, cold SQLite removal and owner restore, reaped successor Claim, forged tokens, corrupt metadata and bounded indexed lookup. A joint initialized catalog/ref base is tested against the reply ceiling. Real workspace tests check certified repository creation and identical custody metadata after fresh-disk restore. These are focused correctness checks, not a full-history or 10,000-developer capacity claim. + +Reproduce the focused checks with the pinned SDK dependencies and Rust 1.98.0: + +```sh +cargo +1.98.0 test -p canopy-server --lib packs::publication --locked -- --test-threads=4 +cargo +1.98.0 test -p canopy-server --lib server::catalog_initialization::tests --locked -- --test-threads=2 +cargo +1.98.0 test -p canopy-server --test multi_server workspace --locked -- --test-threads=4 +cargo +1.98.0 clippy --workspace --all-targets --locked -- -D warnings +cargo +1.98.0 build -p canopy-server --bin canopy --locked +``` + +The `1a11162` owned service checkpoint passes 286 publication and nine workspace/lifecycle cases, all-target workspace Clippy with warnings denied and the server build on macOS. The publication suite includes registrar loss before submission, after acceptance and after a panic, cancellation/closed-service recovery, and restored renewal preserving an existing resolver fence. Frozen Rust-source hashes and protected-checkout/dependency checks accompany the validation. Linux/provider CI and the complete runtime/capacity campaign remain release gates. + +The owner-fencing checkpoint passes 289 publication and nine real startup/workspace cases on frozen macOS/Rust 1.98.0 source, with warnings-denied workspace/all-target Clippy, the server build, formatting and static protection checks. Three new regression families use real durable ownership restoration and missing/corrupt Control objects in both formats. They preserve original committed receipts, reject previous-owner custody, retain permanent shared fences and permit a registered current-owner Claim to restore a usable session. The original failing cold-renewal log is retained. Production mixed-load cost and complete owner-loss/cold staging lifecycle qualification remain open. diff --git a/docs/design/file-attribution.md b/docs/design/file-attribution.md new file mode 100644 index 00000000..6380c1f8 --- /dev/null +++ b/docs/design/file-attribution.md @@ -0,0 +1,77 @@ +# Directory entry file attribution + +Canopy should show the last commit that changed each directory entry at the selected revision. Return the directory page immediately and fill attribution asynchronously from an immutable, commit-specific cache. Use bounded native Git history queries for cold misses; add a persistent index to share unchanged attribution between commits as measured demand warrants. This is a proposed browser implementation, not an implemented endpoint or a production performance result. + +## Current behavior and exact meaning + +`git_read/browse.rs::browser_tree` returns up to 32 entries containing raw names and paths, kind, mode and OID, plus the selected commit. The selected commit is not each entry's last-changing commit. `browser_history` follows first parents and cannot serve as the attribution oracle without changing the semantics below. + +Define the initial algorithm version by the result of this invocation against an authorized, certified snapshot: + +```text +git --no-replace-objects --literal-pathspecs log -1 --format=%H -- +``` + +Pass arguments directly, preserving raw path bytes; never concatenate a shell command. Disable ambient Git configuration and graft/replacement behavior in the managed native workspace. This is path history without rename following. An entry changes when its kind, mode or object ID changes; a directory changes when its subtree changes. Renaming creates attribution at the new path. Deleting and later adding identical bytes counts as a new change. Author timestamps do not define the last change. + +Git's default path history uses parent comparisons and history simplification at merges. A merge identical to a parent can inherit that parent's history; a merge resolution different from every parent is itself a change. Preserve ordered parents. Never silently substitute first-parent integration history. See [Git history simplification](https://git-scm.com/docs/git-log#_history_simplification). + +Display the commit's author, subject, time and a link to that commit. A raw Git author identity is not proof of a Canopy account or the authenticated actor that pushed it. Author-to-account decoration must preserve that distinction. + +## Read protocol + +Resolve the requested ref once to commit C and a certified read generation. Every directory and attribution result carries C. A moving ref must not mix attribution from another revision into the same page. + +Keep the existing directory pagination and expose a batch attribution request for at most one page of paths. Use repository identity, object format, C, literal raw path, and algorithm version as the logical cache key. A blob OID or Git tree OID alone is insufficient: identical content can occur in different histories. + +The response contains the selected commit, one state per requested path, and a dictionary of commit summaries keyed by commit OID. States are `ready`, `pending`, or an explicit unavailable/error disposition. Never fill a missing result with the selected commit. A response can include ready entries while others remain pending. Use a bounded request deadline and an existing bounded polling mechanism or bounded retry token; do not introduce an unbounded job registry or a stream held indefinitely. + +The UI renders names and file actions immediately, reserves space for attribution, then fills author/subject/time together. Discard responses whose commit or page no longer matches the visible page. Coalesce requests for the same commit and directory so many users do not launch duplicate history walks. + +Authorize every request, including cache hits. Acquire the selected certified generation and retain its objects/workspace pins until native work and response ownership finish. Caching attribution neither grants Read nor proves object reachability. A detached observer cannot release resources still owned by a worker. Recheck current access according to the browser read contract before returning results. + +## Cold computation and acceleration + +Initially, use bounded native per-path queries inside the admitted repository read service. Bound workers, queued paths, output bytes, scratch and subprocess lifetime; schedule fairly between repositories and accounts. Do not spawn 32 unconstrained processes for a directory. Cache successful immutable answers and coalesce concurrent misses. Permission failures and budget exhaustion are not history results. + +Generate verified commit graphs with changed-path Bloom filters in background maintenance of certified native workspaces. Bloom filters help history traversal reject commits that did not touch a path; a positive filter result is not proof of a change. They do not guarantee constant-time cold answers. See [Git commit graph maintenance](https://git-scm.com/docs/git-commit-graph). + +Do not replace per-path queries with one naive `git log -- pathA pathB ...` and assign commits from that stream. History simplification for a union of paths can differ from simplification for each individual path. A shared custom walker requires differential qualification per path, including merges. + +Prewarm the default branch's root page and recently viewed directories after publication. Full imports and pushes must not wait for attribution of every file or every historical commit. Under overload, retain immediate directory browsing and return pending attribution. + +## Persistent index and shared data structures + +For large sustained workloads, add an immutable derived index. Its commit binding identifies repository, format, algorithm version and C. An entry contains the exact path, kind/mode/OID and last-changing commit OID. Store commit summaries separately to avoid repeating author and message bytes for every file. + +Reuse the packed architecture's `IndexKey`, `IndexRecord`, `RangeIndex`, `NodeRef`, bounded codecs, verified artifact transport and path-copy updates. Add distinct attribution codec domains and byte-ordered path keys; do not manufacture object IDs from paths. Qualify variable-length keys, node byte limits and wide/deep directories before adopting the generic index. Long keys may require different fanout bounds from object-directory records. A tree node's integrity does not establish that its attribution is semantically correct. + +The proposed recurrence for a present path is: if its exact entry matches a parent, inherit the first matching ordered parent's last-change record; otherwise record C. A root records itself. For directories, compare the subtree entry. This recurrence matched a finite native Git experiment, but remains subject to broader differential qualification before becoming the serving algorithm. Parent order, unusual histories and supported Git versions are part of that qualification. + +Share nodes only when their attribution contents match. Matching Git content trees alone cannot justify sharing history-dependent attribution. An incremental single-parent update should write changed records and their ancestor index nodes, rather than copy all repository paths. Merges require comparison with all relevant parents; their work is not necessarily proportional only to a first-parent diff. Budget and measure merge work independently. + +Use bounded commit-to-attribution-root indexes in immutable artifacts rather than a Cellule SQL row for every file at every commit. Cellule coordinates authoritative repository/catalog facts and, if needed, a bounded descriptor for a published derived index. It does not compute history inside a transaction. Keep attribution replaceable and disposable; it must not delay ref acceptance. + +Represent incomplete coverage explicitly. A missing parent index triggers a bounded history fallback or deferred computation, never inheritance from an incomplete record. Start with per-directory cache coverage and measured hot revisions. Add complete historical roots only through admitted backfill. Avoid accumulating a chain of deltas that every directory request must replay. + +Derived cache retention has explicit quotas and eviction. It must not keep all old commits alive by accident. If attribution artifacts become durable/shared, register their typed storage ownership and retention under the final artifact/GC design; do not use an unrelated staging namespace or historical receipt as a GC root. Rebuilding from certified Git history remains possible after cache loss. + +## Implementation sequence and acceptance + +1. Add a typed attribution record and native history helper, with the exact algorithm version above. Differential tests cover both object formats, empty commits, ordered merges, identical parents with different histories, conflict resolution, octopus merges, modes, symlinks, gitlinks, renames, reverts, delete/re-add, raw non-UTF-8 paths and literal pathspec characters. Include generated DAGs and skewed clocks. +2. Add the commit-pinned batch endpoint, byte/count limits, fair read admission, coalesced cache fills and certified-generation ownership. Test access revocation, force pushes, concurrent pagination, cancellation, timeout, worker death and cache eviction. HTTP history work must stay outside Cellule commands. +3. Add asynchronous directory row decoration. Test navigation during pending requests and partial batch completion. Confirm file browsing works while attribution is backlogged. +4. Measure native fallback and warm cache behavior on Tokio, then full Linux, Kubernetes and Chromium histories. Record cold versus warm storage, path count, history depth, merge shape, native CPU/RSS, queue latency and artifact I/O. Do not reuse line-blame benchmarks as evidence for this feature. +5. If native fallback and page caching miss the workload targets, implement and differentially qualify the shared persistent index. Verify bounded update/read amplification, incomplete backfill, process loss, integrity rejection and rebuild after deleting the cache. Persist only validated answers. + +The proposed warm attribution batch target is p95 below 100 ms. It is an engineering target, not a current guarantee. Keep the directory listing latency independent of attribution history depth. Report cache hit rate, oldest queued job, cache-fill latency, budget refusals and index lag alongside request percentiles. + +Capacity qualification must include 10,000 engineers making 10 commits each in an eight-hour day: 100,000 commits/day, about 3.47 commits/second on average, plus measured bursts. This is not the attribution read rate; model concurrent browsers, pages per session, cache locality and cold misses separately. Require bounded backlog, fair service, stable memory/disk use and foreground push/clone performance while attribution and maintenance run. + +## Evidence and remaining work + +The local design experiment used Git 2.50.1 and compared 101 present file/directory paths across 17 commit states. It passed for its tested root, empty commit, merge, mode, revert, delete/re-add, rename, octopus and literal-path scenarios. The experiment is finite evidence for the proposed recurrence, not a proof for all Git histories or a Canopy API benchmark. + +A separate local Tokio measurement queried 25 root-directory entries at commit `5d5cd8b5b896796445920b3b78c1ad5f9b853fc6` using four bounded native workers with a verified commit graph enabled. Three batch runs took 233.118, 146.671 and 140.715 ms; the median was 146.671 ms. The OS caches were not flushed and the host was shared. These are native fallback measurements, not cache-hit, HTTP, cold-storage or large-team results. + +The production helper, endpoint, UI, cache/index and large-team qualification remain to be implemented. The storage cutover's certified readers, ownership and final retention model are prerequisites for serving this feature through the new architecture. diff --git a/docs/design/final-publication-lifecycle.md b/docs/design/final-publication-lifecycle.md index ddd15c65..01c150a4 100644 --- a/docs/design/final-publication-lifecycle.md +++ b/docs/design/final-publication-lifecycle.md @@ -16,8 +16,10 @@ Retrieve the producer's StagingTask result before waiting for publication. Await ```rust,ignore let base = Arc::new(stage.open_base(indexes, files).await?); -let work = stage.spawn_bound(move |_| async move { +let work = stage.spawn_bound(move |_, context| async move { + context.ensure_live()?; // Build/verify with existing admitted native/disk/reader primitives. + // Retain context ownership in physical jobs that can outlive this future. // Return a private ready_push, ready_root_push or ready_compaction value. prepare_ready(base).await })?; diff --git a/docs/design/initial-preparation-receipts.md b/docs/design/initial-preparation-receipts.md new file mode 100644 index 00000000..6e816437 --- /dev/null +++ b/docs/design/initial-preparation-receipts.md @@ -0,0 +1,31 @@ +# First catalog preparation admission + +A lost Begin reply must not erase knowledge of an accepted preparation when the SDK mutation identity expires or the process loses its local database. `PreparationAdmission` retains the first accepted Begin result independently of the operation's current custody. This is an incremental part of the hard cutover; it does not close pre-dispatch, denied Begin or subsequent Claim/Renew recovery. + +## Shared bounded representation + +Preparation and [staging admission](initial-staging-receipts.md) use one private `InitialAdmission` record, authenticated envelope, query, original-result lookup and restart-token verifier. Each kind has a private column, purpose and typed result validator. Staging retains its existing representation and exact admitted-sequence check. Preparation uses `pushes.initial_preparation`, bounded to 1 KiB. No table, durable queue or per-object metadata is added. + +The record binds tenant/application, exact Begin request, actual admitted SDK identity/digest and `Recorded` reply/commit sequence. The receipt sequence comes from the executing command, rather than being inferred from the attempt token. Begin can observe an existing preparation or bound staging attempt, so its receipt can have a later sequence than the attempt. Different SDK identities observing that same attempt cannot settle their results using the first command's receipt. + +The receiver writes the first receipt after domain admission in the same Cell transaction. Namespace allocation, operation/pin writes, receipt encoding and SDK acceptance either commit together or roll back together. Immutable SQL guards prevent changing, deleting or replacing the initial record. Subsequent Begin, Claim, renewal, completion and reaping preserve first-admission knowledge. Two kinds can coexist in the same logical request row without replacing each other's receipt. + +Admission-only rows remain compatible with pristine initialization and matching compaction requests. Terminal or conflicting outcomes still refuse publication/recreation. That checkpoint used Begin 11/2, Claim 12/2, compaction 22/2 and initialization 31/3. The newer [custody journal](durable-custody-command-intents.md) unbinds raw Begin/Claim/Renew from production, while reusing their domain logic inside commands 41/42. Raw qualification Claim 12 now uses codec 3 for historical journal-grant restart; it is not a production fallback. The shared registry derives descriptors from these typed commands; no old contract is retained for compatibility. + +## Knowledge and custody + +`PreparationAdmission::load` performs an indexed logical request lookup on an authoritative durable Cell head. MAC, purpose, target, actor, operation and digest must match. Corrupt or mismatched metadata is an error, never absence. `original` additionally requires the exact original SDK stamp/digest and incarnation. Its result uses the original receipt even after SDK expiry and fresh-owner restore. These trusted service queries grant no product Read, upload or write permission. + +Recorded timestamps never refresh a session deadline. Current Write, exact operation/pin, expiry and a fresh query remain necessary for preparation factories. Startup compares the recorded owner with its actual current maintenance fence before considering reuse. If that fence still matches, it queries fresh custody at the original receipt watermark and checks the original token/base/format before opening the ordinary resolver. + +The historical base descriptor inside a receipt is evidence of the original reply, not a new artifact retention root. Current independent pins determine preparation retention. Once an attempt is reaped, receipt lookup does not download the old base, and restart Claim selects the current catalog. Typed collection and backup must distinguish this historical knowledge from live custody; retaining every old base merely because its reply is retained would defeat generation reclamation. + +When no final initialization has been registered, startup looks up first-admission knowledge before another Begin. An expired or previous-owner original can be explicitly claimed. If its operation was reaped or aborted, Claim authenticates the exact old receipt and token, checks current Write, logical availability and both quotas, then allocates a new namespace/pin under the actual executing fence and sequence. It selects the current catalog floor and never restores the old pin or grants access to expired artifacts. A different active successor refuses an old Claim. Late restart SQL failure leaves the original Claim's SDK resolution absent and all allocation/custody writes rolled back. + +## Qualification and remaining work + +Seven SHA-1/SHA-256 test families cover actual receipt persistence, distinct receipt/attempt sequences for an already-bound staging attempt, first-result immutability, late insert/update rollback and exact retry, cold restore after deleting SQLite, real SDK expiry, actual new-owner Claim, reaping, forged restart tokens, completed-request refusal, Write revocation, expired custody, corrupt MACs, cross-purpose metadata and absence of invented knowledge for denied/unexecuted Begin. Actual production HTTP creation and fresh-disk restore additionally retain the identical bounded admission record alongside certified initialization and terminal retirement. + +The first accepted Begin record is not an intent journal. It cannot reconstruct an unexecuted or denied command after its original prepared identity is lost. Raw competing Begin identities and later Claim/Renew still need durable original snapshots and results beyond SDK expiry. A successor Claim with a lost reply is not resolved by this initial receipt; startup fails the old-token Claim rather than attributing that successor to the original Begin. Complete registered custody-command recovery, orphan service reconstruction and retained-input adoption remain next work. Production producer/reader conversion, final schema removal, typed collection/backup/isolated restore and full repository/team capacity qualification remain mandatory before publication. + +The later custody journal supersedes first-positive-only startup recovery with original pre-dispatch snapshots and accepted/denied phases, including successor Claim/Renew. This first-admission representation remains in the domain and staging qualification consumers until their conversion; it should be removed rather than retained as a redundant second architecture. See the [journal contract](durable-custody-command-intents.md) for current integration and remaining release work. diff --git a/docs/design/initial-staging-receipts.md b/docs/design/initial-staging-receipts.md index 1cef7e25..a288a157 100644 --- a/docs/design/initial-staging-receipts.md +++ b/docs/design/initial-staging-receipts.md @@ -6,6 +6,8 @@ A Begin acknowledgement can be lost even though its input lease committed. The S The first accepted Begin stores a purpose-specific authenticated record in `pushes.initial_staging`. This reuses the existing logical request row, `CertificateEnvelope`, admitted mutation `Stamp` and `Recorded` result codec. It adds no queue, table or per-object metadata. The blob is at most 1 KiB and contains the tenant/application, exact Begin request, actual SDK identity/digest, original sequence and granted result. The original lease token includes the admitting incarnation, owner fence, attempt and creating namespace. +The private record, query, MAC decoding, original-result lookup and restart-token verifier are now shared with [first catalog preparation admission](initial-preparation-receipts.md). Each kind retains a separate purpose, column and typed result validator. Staging's original wire representation and admitted-sequence rule are preserved. + The receiver records the result in the same Cell transaction that allocates the namespace and inserts the operation and independent lease. A late encoding or SQL failure rolls back all of those writes and SDK acceptance. SQL guards retain the first receipt and forbid mutation, replacement or deletion. Ordinary expiry/reaping may remove custody rows without removing admission knowledge. Root completion updates this pending logical request row using its actor/digest and null completion fields as conditions. It preserves admission metadata while recording the terminal selection. Joint and ref-free completion share the existing result builder. A matching admission-only row does not conflict with compaction, and admission-only rows do not make an otherwise empty repository non-pristine for initialization. Completed or unrelated outcomes retain their existing conflict checks. @@ -25,3 +27,5 @@ After recovering a known grant, the existing supervisor queries live staging cus The original failing test confirms acceptance and a still-live artifact lease, waits for real SDK identity expiry, then recovers the original Begin. Lost acknowledgements and post-execution panics cover both Git object formats. Additional tests exercise final-write rollback and exact retry, local SQL destruction and fresh-owner restore, independent Claim namespaces, lease reaping, Write revocation, immutable rows, mismatched SDK identities, corrupt MACs, forged restart tokens, rollback of a restarted Claim, completed-request refusal and retained uncertainty/credits. The full goal remains open. This record preserves the first accepted initial Begin. Denied Begin commands, competing/repeated raw Begin identities and subsequent Claim/Renew receipts still need the complete registered attempt journal for recovery beyond their SDK expiry. No pre-admission command snapshot is yet discoverable after process loss. Mandatory registration must remove raw unregistered execution paths. Retained input adoption/repreparation, production producers/readers, the fresh-schema hard cutover, full typed GC/backup/restore, continuous maintenance and full-history mixed-load qualification remain required. These receipt tests do not prove the large-team capacity target. + +The local [custody journal](durable-custody-command-intents.md) now provides original pre-dispatch snapshots and positive/negative phases for staging Begin/Claim/Renew/Bind. Production has unbound the raw custody commands; this first-positive-only carrier remains in explicit domain/service qualification consumers pending conversion. Convert those consumers to the journal and remove the redundant carrier before the complete hard cutover is published. diff --git a/docs/design/mandatory-publication-registration.md b/docs/design/mandatory-publication-registration.md index 6fd2a7c2..5eb6f150 100644 --- a/docs/design/mandatory-publication-registration.md +++ b/docs/design/mandatory-publication-registration.md @@ -1,33 +1,47 @@ # Mandatory publication registration -Status: receiver and registered-admission increment qualified locally on 2026-10-03. All existing publication callers in the qualification fixtures now use exact registration. This contract defines the required hard cutover; production startup, producers, readers and fresh-schema conversion remain incomplete. +Status: policy/root registration is published through PR #33. The unpublished production cutover extends this same protocol to final catalog initialization. Production producers, readers, complete startup recovery and final schema conversion remain incomplete. ## Receiver contract -Commands 33, 36 and 38 must find an authenticated recovery record on their exact preparation pin before executing their domain action. The record must match the actual SDK mutation stamp, repository check, tenant, application and incarnation. A missing registration, a competing SDK identity, or a premature frozen refusal returns `NotStarted`, leaves SDK resolution `Absent`, and changes neither domain state nor the registration journal. Retrying the same original command after registration is allowed. +Commands 31, 33, 36 and 38 must find an authenticated recovery record on their exact preparation pin before executing their domain action. The record must match the actual SDK mutation stamp, repository check, tenant, application and incarnation. A missing registration, a competing SDK identity, or a premature frozen refusal returns `NotStarted`, leaves SDK resolution `Absent`, and changes neither domain state nor the registration journal. Retrying the same original command after registration is allowed. Register the exact command body and SDK snapshot before submission. Reuse `Record`, `Bundle`, `SavedCommand`, `Journal`, `Frame`, immutable input roots and `catalog_leases.recovery`; no new durable queue or per-object table is needed. Each policy bundle contains its original page command and one pre-frozen refusal command. Successful pages share that refusal. Advancing the pin requires the authenticated settled predecessor, with strictly increasing steps and a known successful page. A refused policy page cannot advance into positive publication. Domain effects, the original result and sequence, and the phase revision commit in the same SDK transaction. A trusted domain denial is durable knowledge; normalize its typed root reply without inventing another receipt. A later SQL or encoding failure rolls back every effect and SDK acceptance. Retry the original prepared command after repair. -Recovery resolves the authenticated journal and original SDK evidence before reopening bodies or checking fresh custody. Only authoritative absence can execute restored original bytes. Positive cold recovery requires valid custody; a frozen refusal retains its refusal-only role and still checks its actual owner, operation, pin, floor and checkpoint. Original known outcomes remain recoverable after SDK expiry or owner loss. Product response streaming separately requires current Read authorization. +Recovery resolves the authenticated journal and original SDK evidence before reopening bodies or checking fresh custody. Only authoritative absence can execute restored original bytes. Positive root/policy cold recovery requires valid custody; a frozen refusal retains its refusal-only role and still checks its actual owner, operation, pin, floor and checkpoint. Original known outcomes remain recoverable after SDK expiry or owner loss. Product response streaming separately requires current Read authorization. An exact registration retry can refer to a predecessor that has already become historical. `settled_frame` must supply artifact storage to `current_journal` so the existing authenticated history reader can recover that predecessor's original journal. Current-head equality alone is insufficient. The reader checks MACs, matching checks and strictly decreasing steps one bounded frame at a time. ## Protocol and admission -Use recovery purpose `canopy.publication-command-recovery.v3\0` and these command codecs: +Use recovery purpose `canopy.publication-command-recovery.v4\0` and these command codecs in the unpublished hard cutover: | Command | ID | Codec | | --- | --- | --- | +| InitializeCatalogRefs | 31 | 3 | | RegisterRefPolicyPage | 33 | 2 | | CompleteRootPush | 36 | 2 | | CompleteRootOutcome | 38 | 3 | -| RegisterRootRecovery | 39 | 3 | +| RegisterRootRecovery | 39 | 4 | +| ReleaseTerminalRecovery | 40 | 2 | Keep the existing record and artifact structures. Do not add a compatibility decoder or an unregistered execution fallback. -Live factories persist their exact bundle, then bind it into `ReadyBoundRecovery`, preserving the original session, shared clock, lifecycle fence and policy intent. Cold registered work uses `ReadyRootRecovery`. Both use the existing fair publication queue. Raw `ReadyRootPush` and `ReadyRefPolicyPage` values have no admission variant or `From` conversion; persist and bind before submitting. Their private factory values remain available for exact registration and refusal composition. The obsolete direct dispatch code and unused per-page refusal-state allocation are removed. The current reservation formula charges two copies of the body, recovery header and optional refusal: 32 KiB for a final root command and 544 KiB for an armed policy page. Uncertain work stays charged through cancellation and service closure. +Live factories persist their exact bundle, then bind it into `ReadyBoundRecovery`, preserving the original session, shared clock, lifecycle fence and policy intent. Cold registered work uses `ReadyRootRecovery`. Both use the existing fair publication queue. Raw `ReadyRootPush` and `ReadyRefPolicyPage` values have no admission variant or `From` conversion; persist and bind before submitting. Their private factory values remain available for exact registration and refusal composition. The obsolete direct dispatch code and unused per-page refusal-state allocation are removed. The current reservation formula charges two copies of the body, recovery header and optional refusal: 20 KiB for initialization, 32 KiB for a final root command and 544 KiB for an armed policy page. Uncertain work stays charged through cancellation and service closure. + +## Catalog initialization recovery + +`ReadyInitialization` derives the private empty proof from its retained `PreparedCatalog`, freezes command 31 and persists the same exact SDK snapshot/body/header before dispatch. Registration command 39 pins `Kind::Initialization` in the existing attempt namespace. Matching original capabilities can bind into the existing fair publication queue. Production repository startup instead retains this same owner through its already admitted, tracked repository transition. Unknown registration never authorizes final execution. + +Pending startup discovers the latest authenticated custody command before constructing another Begin identity, then queries the current indexed operation/pin binding separately. Its tracked transition can now retire its expired unresolved initialization head through the separate authenticated stop receiver. An observed stop permits a new registered Begin for a wholly absent operation, or Claim of the matching current initialization attempt, without inventing an outcome for the original. See [production initialization after logical retirement](durable-custody-command-intents.md#production-initialization-after-logical-retirement). A recovered positive verifies the original empty catalog/directory/ref roots. Only a known original Stale/Expired final denial permits Claim of that observed attempt; other uncertainty propagates. Ready restoration observes the retained initialization fact without creating a new attempt. + +The preceding first-admission increment recovered the [first accepted preparation admission](initial-preparation-receipts.md) before another Begin. Its original receipt is durable independently of SDK expiry, while current owner/custody are checked separately. The later local [custody intent protocol](durable-custody-command-intents.md) supersedes this startup lookup with a registered original snapshot and positive/negative phase journal. The first-admission carrier remains in domain/qualification consumers pending their conversion. + +Cold initialization performs no new preparation or native work. After authoritative SDK absence it restores the exact original bytes and lets the final receiver atomically check actual owner, Admin, live pin, certificate/checkpoint and pristine roots. Requiring a fresh Write-dependent session first would prevent an expired or revoked original from recording its definitive denial. Live bound dispatch still checks its original shared clock/fence. Known journal outcomes retain their original sequence and receipt even after SDK expiry, owner loss, permission revocation and body loss; they grant no current write or read capability. + +This closes final-command registration and reconstruction only. Initial Begin/Claim/Renew now have a registered snapshot/result protocol in production initialization; the staging/publication services and complete pre-final orphan recovery still need conversion. The local typed [terminal retirement protocol](terminal-publication-retention.md) now verifies its complete empty catalog/directory/ref graph and moves the same original certificate/journal/release receipt into the shared immutable archive before deleting the exact pin. Positive startup discovers the original closed attempt through the immutable initialization fact’s exact pin identity and retires it before serving. A denied initialization retires only after its active binding closes; a successor keeps its independent pin. Unknown attempts remain protected. Background-service reconstruction of older orphan attempts remains part of complete startup integration. Include the immutable initialization roots, command metadata and receipts in typed collection, backup and isolated restore. ## Remaining implementation sequence @@ -63,4 +77,4 @@ The converted late-cursor test requires the actual SQL fault message, SDK absenc Final frozen-source macOS ARM64 qualification passes **655 unique Rust tests**: 531 library tests (6 Git-format / 14 object-storage / 511 server), 104 multi-server tests and 20 CLI/contract/recovery/Smart HTTP tests. The server library finishes in 235.55 seconds and multi-server in 376.05 seconds. Subprocess summaries and focused reruns are excluded from the count. All eight isolated RustFS compatibility cases pass, including SHA-256, signed pushes, SSH, the 4,096-ref mirror, filtered clones and LFS. The separate large-transfer case still requires a dedicated disk with at least 40 GiB free. All 96 Python qualification tests pass in 40.924 seconds. Workspace/all-target Clippy passes with warnings denied in 24.00 seconds. The server binary builds successfully in 80 seconds. Formatting, diff checks, all 409 frozen Rust-source hashes, the protected checkout index, archived document and exact Cellule pin checks pass. Temporary probes are removed. No stack, lifetime, deadline, resource or capacity thresholds were widened. Exact-head Linux/provider CI and full-scale qualification are separate gates. -The published listener fix is independent: PR #32 was merged on 2026-10-03 at `e0957301fa388a69f669cffb31b2126cd982f34d`. Both complete Verify runs passed on its published head `370408f8184c2d0fccb273b6fe6b60ed896123d1`. The fetched `origin/main` and that head have the identical complete tree `45e96c20abad39df2d40c7554c5de2b720ab0dae`; the squash merge therefore includes every published change. The isolated working branch is aligned with that main revision, preserving the local registration increment. There is no open PR #32 conflict to resolve. Those CI checks qualify the published listener tree, not these local registration changes. This increment is prepared on `codex/mandatory-publication-registration` for a new PR to main. The full implementation and capacity goal remains open. +The published listener fix is independent: PR #32 was merged on 2026-10-03 at `e0957301fa388a69f669cffb31b2126cd982f34d`. Both complete Verify runs passed on its published head `370408f8184c2d0fccb273b6fe6b60ed896123d1`. The fetched `origin/main` and that head have the identical complete tree `45e96c20abad39df2d40c7554c5de2b720ab0dae`; the squash merge therefore includes every published change. The isolated working branch is aligned with that main revision, preserving the local registration increment. There is no open PR #32 conflict to resolve. Those CI checks qualify the published listener tree, not these local registration changes. PR #33 merged at `9438bb865959fb975d5349ba8b9908b461653821`. Both complete exact-head Linux [push](https://github.com/crabbuild/canopy/actions/runs/37164049029) and [PR](https://github.com/crabbuild/canopy/actions/runs/37164077931) Verify runs pass at `32f5559216432c0437ac3e864d71c454ee779e1e`, including 659 unique workspace tests, eight RustFS cases and build. Main and that published head have the identical tree `aacb41ee48e953cf106c75f8319667fa83639b96`. The local production cutover starts from that merged main; its changes require their own qualification. The full implementation and capacity goal remains open. diff --git a/docs/design/resident-publication-recovery.md b/docs/design/resident-publication-recovery.md new file mode 100644 index 00000000..ed5ae9d4 --- /dev/null +++ b/docs/design/resident-publication-recovery.md @@ -0,0 +1,150 @@ +# Resident publication recovery ownership + +The selected production repository manager owns recovery services for each locally +resident, certified repository. One node-wide publication budget and one read-round +budget are shared across those owners. This connects the existing exact publication +and retirement primitives to real startup, eviction and shutdown. It is not the +completed producer/reader storage cutover or a repository/team capacity result. + +## Ownership and startup + +`RepositoryManager` creates one `PublicationBudget` from the existing default +publication limits and one `RecoveryScanBudget` with eight read-round permits, +using the same node `TaskTracker` as admitted requests and residency transitions. +Every local `RecoveryServices` constructor receives clones of those budgets. +Scanner constructors require `RecoveryScanSettings`; they cannot create an +independent read budget implicitly. Settings use the immutable directory owner +for account bookkeeping, not a caller-supplied viewer identity. Admission is not +Read, Admin, owner-fence or artifact-retention authority. + +Only after identity and certified catalog initialization succeed does the local +loaded entry start a `RecoverySupervisor::start_retiring` and `CustodySupervisor` +with the actual repository target, artifact store, node preparation authority and +fresh owner fence. The loaded entry retains both scanners and their coordinator +before exposing a serving route. Remote routes do not start recovery scanners. +If the second constructor fails, startup explicitly joins the first worker before +returning the error or relinquishing the residency transition. + +Discovery reuses indexed keyset paging of independent recovery pins and pending +custody heads. Root recovery reconstructs registered exact commands. Custody +retirement marks expired authentic originals logically stopped without inventing +an execution result. Neither scanner retries arbitrary original mutations from a +positive execution phase or substitutes a new operation/request identity. See +[registered recovery](mandatory-publication-registration.md), +[custody retirement](durable-custody-command-intents.md) and +[terminal retention](terminal-publication-retention.md). + +## Bounded reads and pauses + +A read round obtains the shared nonwaiting global/account admission before its +query, metadata reads and visits. The default node cap is eight rounds and the +existing account admission grants at most half that cap to one account. Each +scanner owns at most one round. Refused rounds record a deferral and retry after +the configured interval; they do not create a semaphore waiter or command copy. +Production pages contain at most 128 keys and have a one-second delay. Invalid +page, interval, budget and account bounds reject before task creation. + +Round admission bounds concurrent recovery work, not provider bandwidth, whole +process memory/RSS, native descendants or complete CPU/I/O fairness. Nonwaiting +account headroom alone does not prove starvation-free service across thousands +of resident repositories. Fair continuous scheduling and capacity qualification +remain required. + +The common `ScanControl` serializes round entry against pause/stop. Pause marks +the owner paused, wakes its interval wait and waits for its current round to +finish. It does not drop a Cell query, authenticated artifact read or visit +future. A round publishes diagnostics before its RAII guard releases; after +pause returns those diagnostics are stable. Resume keeps the same worker, cursor +and cumulative diagnostics. A stop is sticky and cannot be undone by resume. +Notify registration precedes state observation to avoid losing a pause/stop +wake-up. Panic/unwind or owned-future cancellation releases the round guard. + +Closing the node scan budget prevents another round and wakes idle/paused +workers. It does not cancel an active round. The node task tracker joins those +workers during shutdown. Scanner failure is logged on join; it does not establish +a command result or authorize discarding uncertainty. Automatic restart after a +scanner panic and process/owner-loss adoption remain separate failure-campaign +work. + +## Eviction and Git maintenance + +The existing per-repository transition guard and bounded residency slots own +release. Candidate selection marks the loaded entry releasing, preventing a +new local fast-path request. Recovery pauses both scanners before checking the +coordinator under its admission lock. + +`close_if_idle` closes only an empty coordinator with no dispatch worker. If any +held, queued, running or uncertain original remains, eviction resumes the same +scanners and restores the serving state. It rejects that candidate and may try +another resident; it never replaces its coordinator or releases its credits. + +An idle owner joins both paused scanners. Its Git maintenance worker has its own +child stop token and retained join handle; eviction cancels and joins it before +Cell release or local directory deletion. A started maintenance round finishes +its owned gateway/provider work rather than being dropped by cancellation. +Idle maintenance does not hold a request pin. + +After those workers quiesce, eviction refreshes the runtime's actual idle Cell +generation: scan reads can invalidate the generation observed during selection. +Only confirmed Cell release permits directory deletion and residency-slot +transfer. Missing/failed release inventory or an ambiguous release retains a +`RefreshHandle` entry. A later load obtains the runtime's actual resident handle, +rebinds the route and starts fresh recovery services; no synthetic capability is +constructed. Cleanup failures retain the released entry and its charged slot. + +## Shutdown and independent progress + +Shutdown closes native admission, stops maintenance, closes scan discovery and +stops HTTP/SSH ingress. After joining ingress it closes and joins the resident +[serving pools](certified-serving-pins.md#bounded-resident-generation-pool) while +publication admission, Cell authority and heartbeat remain available. Borrowed +snapshot clones and detached physical workers retain their exact roots until +authenticated release. Only then does it close/join the node task tracker, +including accepted requests, residency transitions and owned scan/maintenance +rounds. The repository manager then closes publication admission and drains the retained +coordinators. All per-repository drain futures run together, bounded by the +existing loaded-residency cap and shared publication/transport budgets. +A producer-held command in one repository must not prevent exact recovery in +another repository. The drain owns these futures directly; it does not detach +another task inventory or invent a durable queue. + +The loaded-resident mutex is also the construction/closing barrier. Shutdown +permanently closes serving construction under that mutex before collecting the +registered pools. New services register their owner before exposing repository +access under the same mutex. An unpublished constructor that loses the race +joins its own pool, scanners and retained originals in its tracked residency +task before returning `CellDraining`. The subsequent node task join therefore +includes its cleanup even though it was absent from the serving inventory. + +Each drain joins its scanners, waits for dispatch workers and schedules recovery +only for their retained uncertain tickets. Known resolution returns the original +receipt and releases its existing reservation. A held final proof belongs to its +producer: shutdown neither activates nor discards it. An unresolvable original +or held proof keeps shutdown pending, Cell authority, advertisement heartbeat and +workspace ownership intact. Observation timeouts do not cancel that drain. + +Only after repository recovery and native resource ownership drain does shutdown +call `node.shutdown`, confirm workspace cleanup, stop renewal and withdraw the +advertisement. There is no timeout that silently releases unresolved authority. + +## Qualification and remaining cutover + +Regression coverage includes the shared control/admission barrier, tracked +shutdown with an owned active round, independent pause/resume of real indexed +root and custody scanners, authentic orphan retirement, production idle eviction +and certified restoration, rejection/join of a constructor paused before service +publication, preservation of busy held originals, and production +shutdown with held and absent/lost-reply/panicked exact renewal commands across +repositories. Both Git object formats are exercised by the production families. +Final-source totals and retained diagnostic logs are recorded in the +[implementation status](../large-repository-implementation-status.md). + +The current selected packed schema still exposes unconverted legacy consumers. +Production HTTP/SSH/generated staging, conversion to the resident serving pool, +all object/ref/graph/browser/policy/check/merge consumers and final DDL removal +must move together. Admitted immutable custody history/exact lookup, retained +physical input adoption, typed GC/backup/isolated restore, OS resource containment, +accelerated reads/physical rewrite, fair continuous maintenance, signed native +completion/cold clone, file attribution and full Linux/Kubernetes/Chromium plus +10,000-developer mixed-load qualification remain mandatory. This local branch is +published in PR #34 and unreleasable until those gates are complete. diff --git a/docs/design/shared-publication-dispatch.md b/docs/design/shared-publication-dispatch.md index 812664a6..d9e4ad18 100644 --- a/docs/design/shared-publication-dispatch.md +++ b/docs/design/shared-publication-dispatch.md @@ -6,21 +6,24 @@ `PreparedCatalog::ready_push` retains the verified catalog and exact command 19 when publishing refs. `PreparationSession::ready_outcome` retains only the admitted session and exact command 19 for refused/empty outcomes; it requires no catalog artifacts. The prepared-catalog wrapper delegates those outcomes to the same session factory and drops catalog ownership from the ready value. See the [outcome-only contract](outcome-only-completion.md). `PreparedCompaction::ready_compaction` retains the verified compaction and exact command 22; it issues the existing maintenance certificate, checks a 4 KiB input envelope and checks the live lease before and after SDK preparation. Both factories perform verification/certification before admission. Raw descriptors and a caller-selected class cannot construct either ready object. -`PreparedCatalog::ready_root_push` composes the private registered-native completion factory with exact command 36. It verifies the 8 KiB input bound and shared live custody before and after SDK preparation, then retains the prepared catalog and original command. `ReadyPublication::RootPush` enters the foreground class with a 16 KiB reservation for the retained and dispatch copies. PreparationSession::ready_root_outcome retains exact command 38 with only the shared session for registered failed/empty native results. Both use the same RootPush variant and 512-byte reply. It reuses the existing bound handoff, class/account scheduling and recovery slots; no independent queue or mutable response identity is introduced. See the [immutable completion contract](immutable-push-outcomes.md). +`PreparedCatalog::ready_root_push` composes the private registered-native completion factory with exact command 36. It verifies the 8 KiB input bound and shared live custody before and after SDK preparation, then retains the prepared catalog and original command. The raw `ReadyRootPush` factory must persist and bind its original before admission as `ReadyPublication::BoundRecovery`; cold registered work enters as `ReadyPublication::RootRecovery`. Both reserve 32 KiB for the original body and recovery header copies. See the mandatory [registration contract](mandatory-publication-registration.md). PreparationSession::ready_root_outcome retains exact command 38 with only the shared session for registered failed/empty native results. Both use the same registered recovery variants and 512-byte reply. It reuses the existing bound handoff, class/account scheduling and recovery slots; no independent queue or mutable response identity is introduced. See the [immutable completion contract](immutable-push-outcomes.md). `Arc::ready_page` retains the original intent/evidence, verified catalog and exact command 33 under a 512 KiB reservation for two 256 KiB encoded copies. Its `PolicyPage` result preserves the original receipt and never becomes a native response. `StagingTicket::register_policy_page` reuses the existing held publication slot as an intermediate barrier: known success resumes Bound, unarmed known refusal fences, and uncertainty blocks final handoff even if historical SQL progress is complete. Current guard queries and final transactional guard checks remain mandatory. See the [paged-policy contract](paged-ref-policy-guards.md). -An armed policy page retains a shared `ready_root_refusal` command as well as its original page. Composition requires the exact session and refusal-only role; failures preserve both inputs. Its wire reservation is 528 KiB. A known page refusal changes the local phase before executing the exact terminal command; uncertainty/panic recovery preserves the phase-specific SDK evidence. Known page success resumes Bound and does not submit that command. Sharing one refusal Arc across pages avoids repeated native report freezing. The same Arc can also enter final RootPush handoff after later policy/write changes. Refusal-only handoff uses local live custody and the final transaction's authoritative checks rather than requiring fresh Write for renewal/observation. Queued checkpoints and owned work still drain. See the [immutable refusal contract](immutable-push-outcomes.md). +An armed policy page retains a shared `ready_root_refusal` command as well as its original page. Composition requires the exact session and refusal-only role; failures preserve both inputs. Its wire reservation is 544 KiB, including the mandatory original refusal registration header. A known page refusal changes the local phase before executing the exact terminal command; uncertainty/panic recovery preserves the phase-specific SDK evidence. Known page success resumes Bound and does not submit that command. Sharing one refusal Arc across pages avoids repeated native report freezing. The same Arc can also enter final RootPush handoff after later policy/write changes. Refusal-only handoff uses local live custody and the final transaction's authoritative checks rather than requiring fresh Write for renewal/observation. Queued checkpoints and owned work still drain. See the [immutable refusal contract](immutable-push-outcomes.md). PreparationSession::ready_inputs retains the existing command 29 and shared bound session for an adopted native input checkpoint. It checks exact scope/format/adoption context and bounded encoding before SDK preparation; the final command still checks MAC, source custody, current permission, owner, pin and expiry. These checkpoints use the foreground queue. See the [checkpoint contract](native-input-checkpoint.md). They establish descriptor retention rather than physical, canonical or ref authority. -ReadyPreparation::claim and PreparationSession::ready_renew retain exact commands 12/13 with bounded requests and fresh post-commit session observations; see the [bound preparation contract](bound-preparation-dispatch.md). They share foreground admission with an 8 KiB reservation. The [bound lifecycle](bound-preparation-lifecycle.md) now schedules renewal automatically; durable takeover reconstruction remains required. +ReadyPreparation::claim and PreparationSession::ready_renew retain original typed custody command 42 and exact registrar 41 with bounded requests and fresh post-commit session observations; see the [bound preparation contract](bound-preparation-dispatch.md). They share foreground admission with a 28 KiB reservation for both originals and their bounded transport/query copies. The [bound lifecycle](bound-preparation-lifecycle.md) now schedules renewal automatically; durable takeover reconstruction remains required. `ReadyPublication` wraps those private factory outputs. `submit` accepts any factory output and returns the same `PublicationTicket`. Admission failure returns the original typed ready value, preserving its mutation identity and wire bytes. Logical IDs are unique across both classes in one coordinator. `PublicationOutcome` distinguishes committed inline push, immutable root push, policy page, compaction, input checkpoint and bound preparation results. RegisteredNativeInputs preserves the original registration outcome and separately reports fresh checkpoint/bound-session custody; a failed observation fences the shared session without erasing a commit. `PublicationError` preserves the corresponding typed Cellule invocation error, evidence and rejected receipt. `PublicationState` includes held, queued, running, uncertain, finished and proven unexecuted discarded states. `ticket.class()` identifies the class. `ticket.response()` accepts only a completed inline push outcome; `ticket.root_response(store)` requires a completed root push and performs a current authorized query at its original receipt before streaming authenticated bytes. Neither a completed DTO nor a caller-supplied root grants access. Other result kinds refuse both response methods; compactions, input checkpoints and bound preparation commands never become HTTP push responses. These APIs replace the previous push-only outcome shape; there is no compatibility adapter. -## Bounded class and account admission +## Repository class and account admission + +These are repository limits. The constructor also requires a shared node budget, +which applies an additional aggregate cap across repository coordinators. | Default | Bound | | --- | --- | @@ -28,7 +31,7 @@ ReadyPreparation::claim and PreparationSession::ready_renew retain exact command | Maintenance operations | Four reserved slots | | Foreground operations | Remaining 28 slots | | Per actor | Eight operations per class | -| Encoded command reservation | 8 MiB per inline push; 16 KiB per immutable root push; 512 KiB per unarmed policy page, 528 KiB per armed page; 8 KiB per compaction, input checkpoint or bound Claim/Renew command | +| Encoded command reservation | 8 MiB per inline push; 32 KiB per immutable root push; 512 KiB per unarmed policy page, 544 KiB per armed page; 8 KiB per compaction or input checkpoint; 28 KiB per bound Claim/Renew command | | Total command-byte budget | 256 MiB | | Concurrent durability waits | Eight | | Maintenance durability waits | At most two | @@ -40,6 +43,59 @@ Each job records its private factory's reservation; mixed foreground checkpoint/ Configuration requires room for another foreground account, nonzero reserved maintenance slots, checked byte headroom, a burst in 1–32, and a valid maintenance concurrency bound. With multiple durability waits, maintenance cannot use every slot. A one-wait profile permits one maintenance wait; fair class starts then share that serialized dispatch slot. Invalid profiles reject before a coordinator is created. +## Shared node admission and transport + +Create one `PublicationBudget` from the existing `PublicationLimits` profile and +pass clones to every node-local `PublicationCoordinator::new(target, limits, +budget)`. There is no constructor that supplies an independent budget implicitly. +The selected production repository manager creates and retains that shared +instance for its resident recovery coordinators. See the +[resident lifecycle](resident-publication-recovery.md). All remaining producer +and reader integration must reuse this owner; a mandatory constructor argument +alone does not establish whole-service integration. + +The node ledger charges the private ready value's account, class and exact wire +reservation after repository admission succeeds. It bounds the sum of held, +queued, running and uncertain originals across repositories. Any node refusal +returns the original ready value without consuming repository credits or +changing the SDK identity. Foreground and maintenance have independent operation +and byte shares. With the default profile, node foreground admission has 28 +slots and 256 MiB minus 32 KiB; maintenance has four slots and 32 KiB. Node +foreground account admission is at most eight; maintenance account admission is +at most two, leaving room for another account even when one actor administers +many repositories. Account maps exist only while charged jobs exist. + +Transport has separate class and account gates. With the default profile, six +foreground and two maintenance dispatches can be active; an account can occupy +at most three foreground and one maintenance gate. An account acquires its own +gate before the node class gate, so its waiting jobs cannot hold global capacity +needed by another account. Both class lanes must have at least two slots; node +maintenance admission must also have at least two operations. Repository profiles +can still serialize local work. Repository FIFO/account rotation and class burst +scheduling remain in the existing queue; node gates provide bounded concurrency +and account headroom, not a global class-burst, CPU-time or I/O-fairness promise. + +The node transport gate is acquired before making the dispatch body copy and +held through the exact invocation/recovery future. Returning an uncertain result +releases transport capacity while keeping original command credits. A known +terminal result or proven held discard drops retained command/proof resources +before releasing node and repository credits. Observer cancellation releases +neither charge. `PublicationBudget::close` refuses new reservations but leaves +gates usable by already admitted activation and exact recovery; it does not +cancel or drain repository workers. `stats` exposes charged class/account/byte +occupancy, acquired class transport gates and admission closure. + +This is resource admission, not a durable outcome owner or artifact retention +authority. The production service must retain coordinators and returned uncertain +tickets, explicitly stop scanners, drain workers and resolve exact originals +before releasing the Cell or deleting its workspace. Idle scanners must not +prevent repository eviction indefinitely. The selected resident recovery owner +now pauses/joins scanners and Git maintenance before release, rejects busy +coordinators without abandoning originals, and drains repository recovery before +node authority/workspace cleanup. Production startup/eviction/shutdown regression +evidence and its limits are in the [resident contract](resident-publication-recovery.md). +The full producer/reader conversion and capacity qualification remain open. + ## Held ownership and fair starts try_reserve admits a charged Held job synchronously without execution. It returns the original ready value on capacity, contention, duplicate, target or closure refusal. activate joins the existing fair queue once; discard_held succeeds only before activation, dropping resources before credits. Both remain usable after close for existing admission. close_and_drain returns held and uncertain jobs still charged. See the [final lifecycle handoff](final-publication-lifecycle.md) for worker/renewal/checkpoint ordering and observation-only final tickets. @@ -50,14 +106,27 @@ The shared queue contains two instances of the existing account-fair queue. With Catalog/ref CAS and current policy/ACL checks remain in the authoritative command. Two preparations against one old catalog can conflict even when both dispatch fairly. Uploads, native decoding, canonical verification and reconciliation never run inside this queue. A known durable catalog conflict may reenter only with a newly prepared command identity and a properly reconciled certificate. -Dropping an observer does not cancel admitted execution. Pending, malformed published and panicked-task outcomes retain the original ready value and reservation. `pending`, `recover` and `close_and_drain` handle both classes. Recovery joins the same class/account queues. Staging Begin/Renew/Bind now reuse this same exact invocation/resolution implementation with a 4 KiB decoded-result bound; inline push and compaction results retain their 128-byte bound; immutable root push and policy-page results use a 512-byte bound. Bound input checkpoints and Claim/Renew commands share this foreground dispatcher with a 4 KiB decoded-result bound and fresh post-commit custody queries. Staging has its own long-input admission/lifecycle rather than entering the final-command fair queues; see the [service contract](staging-service-lifecycle.md). Inline push, immutable root push, policy-page and compaction dispatch check local session custody before initial submission and after authoritative absence. Resolve a known committed outcome before that guard; decode a committed result with its original receipt without rerunning its handler. Unknown, expired, unreachable or changed-incarnation evidence remains uncertain. Never replace its proof or mutation identity while acceptance is unknown. +Dropping an observer does not cancel admitted execution. Pending, malformed published and panicked-task outcomes retain the original ready value and reservation. `pending`, `recover` and `close_and_drain` handle both classes. Recovery joins the same class/account queues. Staging and bound custody commands now use mandatory registered-original recovery with metadata-first outcomes and retained registrar identity; inline push and compaction results retain their 128-byte bound; immutable root push and policy-page results use a 512-byte bound. Bound input checkpoints and Claim/Renew commands share this foreground dispatcher with a 4 KiB decoded-result bound and fresh post-commit custody queries. Staging has its own long-input admission/lifecycle rather than entering the final-command fair queues; see the [service contract](staging-service-lifecycle.md). Inline push, immutable root push, policy-page and compaction dispatch check local session custody before initial submission and after authoritative absence. Resolve a known committed outcome before that guard; decode a committed result with its original receipt without rerunning its handler. Unknown, expired, unreachable or changed-incarnation evidence remains uncertain. Never replace its proof or mutation identity while acceptance is unknown. A terminal result drops dispatch/retained proof ownership before releasing class/account/byte credits. Resolved tickets retain only bounded result/read context. Recovery remains possible after closing admission. The bound lifecycle now owns automatic renewal and bound Claim; accepted input registration does not renew a lease or extend the original generation floor. The coordinator is service-owned local state, not a durable outbox or permission to delete remote inputs. ## Integration and evidence -Keep one coordinator and geometric planner per repository. Obtain a fresh admitted query-derived maintenance base, call the [geometric planner](geometric-directory-maintenance.md), wrap the verified result in an Arc and call `ready_compaction`, then `submit`. Observe or recover the exact ticket before releasing uncertain inputs. Obtain a fresh frontier for the next preparation. Integrate process admission, fair CPU/I/O shares, renewal/reaping, owner-loss reconstruction and complete retained-root inventory before selecting production handlers. +Keep one coordinator and geometric planner per repository, with one shared publication budget owned by the node. Obtain a fresh admitted query-derived maintenance base, call the [geometric planner](geometric-directory-maintenance.md), wrap the verified result in an Arc and call `ready_compaction`, then `submit`. Observe or recover the exact ticket before releasing uncertain inputs. Obtain a fresh frontier for the next preparation. Integrate process admission, fair CPU/I/O shares, renewal/reaping, owner-loss reconstruction and complete retained-root inventory before selecting production handlers. Tests exercise class/account admission, retained failure values, duplicate logical IDs, maintenance concurrency while foreground completes, canceled observers, current admin revocation, and SHA-1/SHA-256 absent/lost-acknowledgement/panic recovery with original receipts and exactly one logical outcome. The existing push dispatcher tests remain in place with typed-result assertions. The geometric native fixture now prepares and publishes repeatedly through this shared dispatcher until ingress and level debt drain, checking canonical/source/version identity, unchanged refs and old-reader access. These fixtures establish bounded dispatch and recovery. They do not establish stable maintenance service under 35 pushes/s, full-history amplification, durability grouping, source-independent restore or capacity for 10,000 engineers. The mandatory workload and recovery campaigns remain release gates. + +## Serving retention release + +`ReadyServingRelease` joins the existing maintenance class under an 8 KiB +reservation, with the original exact 1 KiB command and 128-byte result. Its +private factory requires sticky serving closure and actual physical read-worker +drain. A distinct typed job kind keeps its real reader ID separate from both +creating publications and custody retirement; it does not fabricate an artifact +namespace or grant preparation authority. Uncertainty retains the original +command/owner/credits, and release recovery is looked up through +`pending_serving_release`. See the [serving contract](certified-serving-pins.md). +Production acquisition/renewal, generation caching and read-owner handoff still +require integration. diff --git a/docs/design/staging-service-lifecycle.md b/docs/design/staging-service-lifecycle.md index f92971f6..86867440 100644 --- a/docs/design/staging-service-lifecycle.md +++ b/docs/design/staging-service-lifecycle.md @@ -4,7 +4,7 @@ ## Admission and ownership -Keep one service-owned coordinator per repository. `ReadyStaging::new` prepares the exact SDK BeginStaging command without executing it; `submit` performs synchronous local admission and starts service-owned supervision. Failure returns the original ready command and reason, so retry preserves mutation identity and bytes. Foreign targets, incompatible lease duration, duplicate logical IDs, closed admission and capacity reject before execution. +Keep one service-owned coordinator per repository. `ReadyStaging::new` prepares the original typed custody command 42 and its exact registrar command 41 without executing either; `submit` performs synchronous local admission and starts service-owned supervision. Failure returns the original ready command and reason, so retry preserves mutation identity and bytes. Foreign targets, incompatible lease duration, duplicate logical IDs, closed admission and capacity reject before execution. | Default bound | Value | | --- | --- | @@ -12,8 +12,8 @@ Keep one service-owned coordinator per repository. `ReadyStaging::new` prepares | Operations per actor | Eight | | Input workers and retained completed results | 64 | | Workers and retained results per actor | Eight | -| Encoded command envelope | 4 KiB | -| Command reservation per operation | 8 KiB for retained and transport copies; another 4 KiB after checkpoint admission | +| Custody envelopes | 1 KiB original execution body; 4 KiB complete registrar intent | +| Command reservation per operation | 28 KiB covering retained/dispatch intent bodies, registrar transport/query decode and original reply/body ceilings; another 4 KiB after checkpoint admission | | Checkpoint slots per operation | One bounded request and retained result | | Renewed input lease | 60 seconds | | Renewal lead time | 30 seconds | @@ -30,13 +30,17 @@ After Begin or Claim succeeds, query CheckStaging at its receipt. Derive the loc The service periodically prepares and executes RenewStaging before that deadline. Each renewal has a fresh mutation identity; an ambiguous renewal retains its original command instead of allocating another. After a known success, another authoritative query checks live identity, format, expiry and current access before advancing the shared deadline. Producers receive StagingContext, which supplies the checked namespace token and format, and observes the shared conservative deadline and lifetime. -`ticket.spawn` owns and supervises the producer future. Its returned StagingTask observes the result. A producer error, panic, expired custody or lost access fences the job. Cancellation aborts and joins the producer before releasing its worker credit. Completed retained inputs drop before their credit on fencing, including when an external observer remains alive. Results transfer once through `wait`; resources move to the caller before that slot releases. Retained results count toward both global and actor worker caps, preventing an unbounded completed-result backlog. +`ticket.spawn` owns and supervises the producer future. Its returned StagingTask observes the result. A producer error, panic, expired custody or lost access fences the job. Cancellation aborts and joins the async producer; its worker credit remains charged until every physical owner also drains. Completed retained inputs drop before their credit on fencing, including when an external observer remains alive. Results transfer once through `wait`; resources move to the caller before the result slot releases its ownership. Transferred physical owners can keep the original worker admission charged afterward. Retained results count toward both global and actor worker caps, preventing an unbounded completed-result backlog. -Use existing admitted native-process, workspace, reader and disk primitives inside producers. The callback counter does not account for arbitrary heap allocation, unjoined descendants or detached physical readers. Their independent admission and retention obligations remain in force; production integration must preserve them through work and handoff. +Each `StagingContext` clone now retains the same original worker/actor admission. The internal `physical_owner` supplies that lifetime pin without granting custody or publication authority. `spawn_bound` supplies both the live shared preparation session and this context. A detached job must retain the pin through its physical completion; dropping an async observer, returning a result, or expiring custody cannot return that worker's capacity early. No new worker slot or queue is created by cloning ownership. + +Native receive requires a live context and retains its owner in the existing Git process group. Response reconciliation and pack enumeration retain it in blocking jobs; capture files retain it through hashing and provider upload. `PhysicalVerifier::download_staged` validates the input's repository, namespace and format against live context, carries the pin through cache creation/download, isolated native verification, canonical extraction and edge-spool writes, and checks custody again before each new inspection and final witness. Deferred workspace cleanup retains the original worker credit. The unowned physical download entry point is qualification-only. + +Use the existing independent native-process, workspace, reader and disk admission inside these producers as well. Worker ownership is a drain barrier, not CPU/I/O fairness or a bound on arbitrary heap allocation. The live HTTP/SSH/generated write factories and resident staging service still need conversion; not all lower request, metadata and policy workers have been connected to this pin yet. ## Durable input checkpoint registration -After sealing a NativeInputCertificate, call `ticket.register_inputs(proof, identity)` before seal or stop. Admission synchronously transfers that bounded envelope and exact mutation identity into one service-owned checkpoint slot. Local checks require an active live stage and matching actor/token/target. A completed successful slot can be replaced while unbound only by a proof naming that exact checkpoint digest, enabling the [request-before-native append sequence](durable-push-request.md). Pending/uncertain/failed slots and unrelated proofs reject; failure returns the original proof without executing. Existing observers retain their original receipt. The slot remains charged while the job is admitted, including uncertainty and retained completion. It adds 4 KiB to the existing 8 KiB command reservation; it cannot form an unbounded queue. +After sealing a NativeInputCertificate, call `ticket.register_inputs(proof, identity)` before seal or stop. Admission synchronously transfers that bounded envelope and exact mutation identity into one service-owned checkpoint slot. Local checks require an active live stage and matching actor/token/target. A completed successful slot can be replaced while unbound only by a proof naming that exact checkpoint digest, enabling the [request-before-native append sequence](durable-push-request.md). Pending/uncertain/failed slots and unrelated proofs reject; failure returns the original proof without executing. Existing observers retain their original receipt. The slot remains charged while the job is admitted, including uncertainty and retained completion. It adds 4 KiB to the existing 28 KiB command reservation; it cannot form an unbounded queue. The same supervisor prepares command 29 and stores its exact SDK command before dispatch. Due renewal precedes queued registration; accepted registration precedes Bind or graceful stop. `StagedInputsTicket::wait` observes the original durable receipt or uncertainty/error. Dropping it never discards the queued or executing command; `pending_inputs` retrieves the observer. `recover(ticket)` resolves the exact registration without replacing identity or bytes. Closing returns uncertain registrations with their existing reservations. @@ -46,7 +50,7 @@ A known registration stores its original receipt before a fresh CheckStaging que Call seal when the input phase should finish. It prevents new producer admission and enters Draining. Existing producers and retained completed results continue under renewed staging custody. Bind does not begin until all input slots have drained through handoff or failure. This prevents a canceled observer from silently losing a physical witness while the service advances to catalog preparation. -BindStaging uses a freshly prepared exact SDK command. Known binding preserves the token, creating namespace and artifact expiry, and adds only the current catalog floor. Bound records that durable result and its original receipt; its recorded timestamps are not a fresh live-lease observation. Stage contexts become inactive after handoff. The operation remains admitted through bound preparation. `ticket.open_base` refreshes at the binding receipt and uses the existing PreparationBaseResolver with the supervisor's shared session, validating current access and expiry while inheriting automatic renewal, shutdown fencing and the bound residence ceiling. +Bind uses a newly prepared original command 42 and registrar 41 under the shared registered custody protocol. Known binding preserves the token, creating namespace and artifact expiry, and adds only the current catalog floor. Bound records that durable result and its original receipt; its recorded timestamps are not a fresh live-lease observation. Stage contexts become inactive after handoff. The operation remains admitted through bound preparation. `ticket.open_base` refreshes at the binding receipt and uses the existing PreparationBaseResolver with the supervisor's shared session, validating current access and expiry while inheriting automatic renewal, shutdown fencing and the bound residence ceiling. A producer can physically verify a native pack and return its private PhysicalPackWitness and sealed metadata segments. Take that result, seal, observe Bound, open the base, and feed the witness/segments to CatalogPreparation. The existing assembler rechecks store, namespace, partition completeness, canonical overlap and closure. Its private factories issue the publication proof. Bind and a generic producer result do not grant canonical or publication authority. @@ -54,13 +58,25 @@ Bound preparation is now automatically renewed by this service, and spawn_bound ## Exact uncertainty and shutdown -Begin, Claim, Renew, RegisterStagedInputs and Bind share the same exact invocation/resolution implementation with push and compaction dispatch. Resolution of authoritative absence permits execution of the retained exact command. A committed outcome decodes with its original receipt; it never reruns the handler. Unknown, expired, unreachable, changed-incarnation and malformed published results retain evidence and reservation. +Staging and bound Begin/Claim/Renew/Bind retain both original command 42 and registrar 41 through the [custody intent protocol](durable-custody-command-intents.md). Registration must be known before original execution; metadata results are observed before SDK expiry or local execution guards. RegisterStagedInputs retains its separate original checkpoint command and exact SDK invocation/resolution path. Resolution of authoritative absence permits execution of the retained exact command only while the local fence, deadline and residence ceiling allow new execution. Known outcomes are returned before that guard. A committed outcome decodes with its original receipt; it never reruns the handler. Unknown, expired, unreachable, changed-incarnation and malformed published results retain evidence and reservation. -Uncertain stops new producer admission. Existing work can continue only through its previously established deadline. `recover(ticket)` resumes the exact retained command; it cannot replace its identity or bytes. No new renewal, registration or bind is issued while an earlier command remains ambiguous. Panicked command tasks retain pending evidence. Unexpected service-worker failure fences local work and requires explicit exact recovery before restarting supervision. +Uncertain stops new producer admission. Existing work can continue only through its previously established deadline. `recover(ticket)` resumes the exact retained command; it cannot replace its identity or bytes. No new renewal, registration or bind is issued while an earlier command remains ambiguous. Panicked command tasks retain pending evidence. Unexpected service-worker failure fences local work and waits for exact recovery before restarting supervision; authenticated retirement can schedule that recovery automatically. `stop` prevents new workers and waits for accepted input tasks/results to drain while renewal continues. It does not retract an independent SQL pin. `close_and_drain` closes all admission, stops jobs and returns still-charged uncertain tickets once running commands and input slots have drained. Service consumers must take retained completed results before a graceful stop can finish; retrieve lost observers through pending_task. Recovery remains possible after closing. A reached lifetime or lost authority fences and discards untransferred results conservatively. -This service is process-local ownership, not a durable outbox or authenticated input inventory after process loss. Owner takeover must resolve exact/logical outcomes and reconstruct or adopt retained physical inputs under the new admitted namespace through the [authenticated input checkpoint protocol](native-input-checkpoint.md). ReadyStaging::claim now retains/resolves the exact Claim command and supplies a fresh staging context. Staging checkpoint supervision now exists; exact checkpoint supervision after bound Claim now uses the publication dispatcher. Production producer wiring, durable takeover reconstruction and complete wire-plan/response recovery remain required. Final publication now uses the existing fair coordinator through an observation-only lifecycle ticket; accepted final intent continues through close, while a pre-activation fence discards only proven unexecuted work. Neither local completion nor SQL reaping authorizes remote deletion. +This service map is process-local. Registered custody intents preserve original command knowledge across process loss, but they do not reconstruct local worker ownership, a fresh current-owner lease or an authenticated physical input inventory. Owner takeover must resolve exact/logical outcomes and reconstruct or adopt retained physical inputs under the new admitted namespace through the [authenticated input checkpoint protocol](native-input-checkpoint.md). ReadyStaging::claim now retains/resolves the exact Claim command and supplies a fresh staging context. Staging checkpoint supervision now exists; exact checkpoint supervision after bound Claim now uses the publication dispatcher. Production producer wiring, durable takeover reconstruction and complete wire-plan/response recovery remain required. Final publication now uses the existing fair coordinator through an observation-only lifecycle ticket; accepted final intent continues through close, while a pre-activation fence discards only proven unexecuted work. Neither local completion nor SQL reaping authorizes remote deletion. + +## Cold custody reconstruction + +`ReadyStaging::restore(client, target, operation)` authenticates and loads the latest registered original without preparing a new original or registrar identity. It supports all seven custody actions: staging and preparation Begin/Claim/Renew, plus Bind. The coordinator resolves the original phase and receipt before checking fresh lease and actual-owner authority. `StagingTicket::restored_evidence` and `restored_outcome` expose historical knowledge, including denials, after local fencing or shutdown; they do not authorize work. Changed restart lease profiles cannot rewrite or hide the original command's duration or result. + +Known staging grants acquire fresh staging custody. Known preparation grants use the same fresh bound-session opener as warm Bind/Claim. Recorded clocks never establish a local deadline. Authenticated frozen commands execute only after authoritative SDK absence; the existing receiver atomically enforces current authorization, token and actual ownership. Old-owner Renew/Bind can settle their original stale denial under the new owner without granting custody. Unknown/expired resolution, unavailable queries and lost replies preserve exact evidence and admission for explicit recovery, including on a closed coordinator. Expired unresolved commands cannot become synthetic denials or fresh retries. + +The shared preparation fence now retains a terminal watch value. A bound worker observes that signal independently of coordinator status changes or renewal timers, aborts and joins its callback, and drops owned results/resources before releasing credit. Late subscribers observe the already-fired fence. Bound context checks and cancellation deadlines include the shared session's live lease and ceiling. This qualifies callback ownership; it does not establish OS containment for arbitrary detached subprocesses or I/O. + +The command-wire reservation remains 28 KiB, or 32 KiB with a checkpoint, and operation/actor/worker admission caps are unchanged. These bounds do not claim total resident heap or Control/advertisement I/O accounting. Cold restore does not resurrect old workers, authenticate a new physical inventory, reconstruct an unregistered registrar, or resolve an expired unresolved original. The [custody retirement service](durable-custody-command-intents.md#separate-retirement-of-expired-originals) now records a separate stop for expired unresolved heads. Explicit exact recovery observes that typed fact, fences/drains and returns local admission while retaining original evidence without an execution reply. The local staging service now automatically observes authenticated retirement as described below. General production scanner ownership/drain, settled-history archival, retained-input adoption and takeover wiring remain required. The separate [production initialization transition](durable-custody-command-intents.md#production-initialization-after-logical-retirement) now retires its own expired unresolved head and chooses an explicit successor; it does not instantiate the general staging scanner lifecycle. + +Seven regression families cover both object formats and all seven command kinds; actual durable owner restore after local SQLite removal; original positive/negative receipts; changed restart profiles; authoritative absence; lost replies/panics/private-query failure; closed-service recovery; expired unresolved originals; and resource drop before credit release when a shared session fences. Their native workers and histories are small fixtures, not a large-team capacity result. ## Evidence and remaining work @@ -69,3 +85,17 @@ Nine service tests cover canceled observers and single typed handoff; operation/ Five additional checkpoint service tests cover canceled observers; absent, lost-reply and panicked exact dispatch in both OID formats; registration-before-Bind ordering and original receipt replay; one-slot and foreign/duplicate/closed refusal without execution; committed recovery followed by current-access revocation; and authoritative expiry after absence. The real receive/publication/cold-clone fixture uses service-owned registration in both formats. These tests establish protocol composition and ownership on small fixtures. They do not establish four-hour full-history throughput, stable maintenance under peak traffic, source-independent restore or capacity for 10,000 engineers. Production producer wiring, complete resource/descendant admission, authenticated durable input inventories and owner-loss adoption, remaining-floor configuration, complete retained-root reclamation, accelerated readers and mandatory mixed-load/recovery campaigns remain release gates. + + +## Automatic observation of custody retirement + +An uncertain registered custody command starts one shared read-only retirement probe for its existing StagingCoordinator. It scans the already admitted map; it creates neither another durable outbox nor one polling task per operation. Default rounds retain at most 128 operation keys/job references and issue one private metadata query at a time. Byte-ordered keyset rotation uses an explicit beginning-of-pass cursor, advances past unavailable/malformed heads and wraps to revisit earlier keys. A round stops selecting further work after its one-second budget elapses, retains ownership until a slow query completes, then waits one second. This budget is not a query deadline or a cold-storage latency guarantee. Unexpected probe-worker failure retains the original jobs and restarts the single probe after the same delay. + +The probe takes a thin exact fingerprint under the job's command lock: target, operation/ordinal, intent digest and original PendingMutation. It does not clone the original command/registrar bodies across its independent query await. Loading uses the existing indexed exact-ordinal lookup and authenticated codec, so a later registered successor cannot hide the old original. Only a valid stop fact for that original schedules existing exact recovery. A missing/private/malformed observation retains admission. A known original execution phase alone does not trigger automatic execution/retry. Checkpoint command 29 and final publication remain under their distinct exact owners and are excluded from custody probing. + +Exact recovery independently reloads/authenticates closure, reports typed Stopped with the original evidence, fences the shared session and joins/cancels callbacks. It drops untransferred completed resources before their worker credit and removes the staging operation only after all workers drain. A stop is logical closure, never a fabricated command-42 execution receipt or denial. This continues after observer drop or coordinator closure. The probe relinquishes ownership when no eligible jobs remain; admission and that ownership change use the same mutex to avoid a lost new-job wakeup. + +StagingStats exposes completed probe queries, failed observations/fingerprint construction, scheduled exact recoveries, worker restarts and whether a probe is running. Counters saturate and remain bounded. The existing 28/32 KiB command-wire reservation and operation/actor/worker caps are unchanged. Private query transport/decode is independently admitted by the SDK; these counters and wire reservations are not total heap, RSS, provider-I/O or latency qualification. This mechanism observes existing authenticated stops; production repository lifecycle ownership of the stop scanner, physical input adoption and cold producer takeover remain required. + + +Seven additional regression families exercise all seven original custody actions in SHA-1/SHA-256, closed coordinators and dropped observers, stopped staging/bound renewals with live callbacks and retained completed resources, unavailable private queries without absent-command execution or known-phase retries, an old ordinal after an explicit successor, malformed stop rejection, a corrupt head followed by a valid head and later repair, exclusion of input checkpoints despite an older stop, and 130 admitted operations spanning multiple probe pages plus restart at an earlier key after idle. The native/domain codecs reject the all-zero operation ID; the multi-page fixture uses valid nonzero IDs rather than weakening that invariant. The initial warm fixture observed the preceding binding before the renewal; it now waits for the actual uncertain renewal. The checkpoint fixture now registers an explicit successor instead of using the fresh-operation factory against an existing journal. Final frozen-source evidence is recorded in the implementation status. diff --git a/docs/design/terminal-publication-retention.md b/docs/design/terminal-publication-retention.md index 5b8ed9fe..6fe01b40 100644 --- a/docs/design/terminal-publication-retention.md +++ b/docs/design/terminal-publication-retention.md @@ -1,22 +1,22 @@ # Terminal publication recovery retention -Completed pushes must release their independent preparation pins without losing original command receipts. Leaving successful pins forever eventually exhausts the 4,096-pin admission cap and retains unnecessary catalog floors. This protocol transfers the same authenticated recovery certificate and phase journal into the existing immutable `pushes` row, then deletes the matching pin in one transaction. It uses the fresh publication schema and existing artifact structures; it adds no durable queue or per-object rows. +Completed pushes and closed initialization attempts must release their independent preparation pins without losing original command receipts. Leaving successful pins forever eventually exhausts the 4,096-pin admission cap and retains unnecessary catalog floors. This protocol transfers the same authenticated recovery certificate and phase journal into the immutable shared `catalog_recovery_receipts` table, then deletes the matching pin in one transaction. It uses the fresh publication schema and existing certificate, journal, SDK receipt and artifact structures; it adds no durable queue or per-object rows. The archive is keyed by original incarnation/admission sequence, with an operation index for typed enumeration. Different attempts of the same logical initialization retain independent original receipts. The protocol releases a preparation pin. It does not authorize provider deletion. Complete retained-root enumeration, reader and worker drain, backup, isolated restore, and repository-scoped collection remain required before any production artifact deletion. ## Release eligibility and authority -`RegisteredRootRecovery::ready_terminal_release` accepts only the current canonical recovery head with a recorded completed root outcome. A policy head is terminal only when its original refusal is recorded and its pre-frozen fallback root command has completed. Unknown acceptance, passed intermediate pages, denied root commands, historical heads, and a refused page without completed fallback cannot produce a release proof. +`RegisteredRootRecovery::ready_terminal_release` accepts only the current canonical recovery head with a recorded completed root outcome. A policy head is terminal only when its original refusal is recorded and its pre-frozen fallback root command has completed. A known initialization result is terminal after its exact attempt is closed. A positive must match the immutable selected initialization fact; a known denial may retire only after Claim or bounded operation reaping has removed that exact active binding. A successor of the same logical operation keeps its own pin. Unknown acceptance, passed intermediate pages, denied push root commands, historical heads, and a refused page without completed fallback cannot produce a release proof. -The private factory checks the saved actor and request selection against that exact completed outcome. It authenticates the current bundle and each predecessor frame, verifying repository MACs, tenant/application, lease identity, original SDK stamps, and strictly decreasing phase steps. Each header is bounded by 8 KiB. It then authenticates the selected outcome/native metadata and streams the selected response and native audit bodies to verified EOF. Missing or corrupt required bytes prevent proof creation. +The private factory checks the saved actor and request selection against that exact completed outcome. It authenticates the current bundle and each predecessor frame, verifying repository MACs, tenant/application, lease identity, original SDK stamps, and strictly decreasing phase steps. Each header is bounded by 8 KiB. It then authenticates the selected outcome/native metadata and streams the selected response and native audit bodies to verified EOF. Missing or corrupt required bytes prevent proof creation. For positive initialization, the same verifier used by route activation downloads the catalog, directory and ref snapshot and checks their complete typed empty graph. The graph transcript includes the original fact and all three authenticated descriptors. A known negative has no positive graph edges; its original typed denial remains bound by the phase digest. -The purpose-specific MAC proof binds the original recovery certificate, phase-journal digest, and verified closed-graph digest. `ReleaseTerminalRecovery` is command 40, codec version 1, with input limited to 4 KiB and output to 128 bytes. The actual command receiver separately requires current repository Admin authorization and its admitted owner fence. A proof prepared under an old owner or by a subsequently unauthorized administrator cannot bypass those checks. +The purpose-specific MAC proof binds the original recovery certificate, phase-journal digest, and verified closed-graph digest. `ReleaseTerminalRecovery` is command 40, codec version 2, using purpose `canopy.terminal-recovery-release.v2\0`, with input limited to 4 KiB and output to 128 bytes. The actual command receiver separately requires current repository Admin authorization and its admitted owner fence. A proof prepared under an old owner or by a subsequently unauthorized administrator cannot bypass those checks. ## Atomic transfer and original receipts -Before its first write, the receiver verifies the proof and embedded original certificate, exact current pin identity/head/phase, terminal selection, immutable saved push outcome, and absence of an active logical operation. All semantic refusals precede writes. +Before its first write, the receiver verifies the proof and embedded original certificate, exact current pin identity/head/phase, terminal selection, the immutable selected push or positive initialization outcome, and absence of an active binding for that exact incarnation/admission sequence. All semantic refusals precede writes. -The receiver obtains its release command's actual SDK mutation evidence and sequence from `CommandContext`. It writes three bounded values to the existing push row: the original recovery certificate, original phase journal, and release identity/result/sequence. An exact CAS then deletes the matching preparation pin. Any later SQL failure rolls back both writes and SDK acceptance. SQL guards prohibit archive replacement, mutation or deletion; the lease deletion guard requires the same certificate and phase in the completed push row. +The receiver obtains its release command's actual SDK mutation evidence and sequence from `CommandContext`. It inserts one shared archive row containing the original pin key and logical operation, and three bounded values: the original recovery certificate, original phase journal, and release identity/result/sequence. An exact CAS then deletes the matching preparation pin. Any later SQL failure rolls back both writes and SDK acceptance. SQL guards prohibit archive replacement, mutation or deletion; the lease deletion guard requires the same certificate and phase in the exact shared archive row. Archived recovery reuses `RegisteredRootRecovery` and the existing predecessor-frame walk. The original root/page receipt survives SDK identity expiry, removal of original command bodies, local SQL destruction and fresh-owner restoration. A completed release also resolves its original archived receipt before SDK resolution or current custody checks. A competing, separately prepared release identity cannot inherit that receipt; once another release wins, its receiver refuses Missing. @@ -26,17 +26,18 @@ An owner SQL query is not an arbitrary local SQLite read. The pinned Cellule exe | Root or edge | Retention after this attempt closes | | --- | --- | -| Recovery certificate and phase journal | Stored unchanged in the selected push row | +| Recovery certificate and phase journal | Stored unchanged in the shared immutable receipt archive | | Original head bundle and predecessor frames/bundles | Required to resolve exact historical command identities and receipts | | Selected outcome root and native result root | Required immutable metadata for replay and audit | | Selected response body | Required; its creating namespace can belong to a prior admitted attempt | | Native plan and signed certificate bodies | Required native audit edges, when present | | Original wire request, original command bodies, unselected responses and unpublished candidate artifacts | No permanent edge from this closed audit role; other retained snapshots, unknown attempts, readers or backups can still require them | +| Initial empty catalog, directory and ref snapshot | Retained through the immutable initialization fact; its original pin identity also names receipt recovery | | Published catalog, refs, packs and indexes | Retained through their own certified catalog/generation and reader/backup roots | `root_completion::closed_graph` verifies the closed audit edges using the existing `StoredInputRoot`, `ArtifactDescriptor` and native-result representations. It retains no whole body in memory; authenticated parts are streamed one at a time. This verifier is not an exhaustive retained-root inventory or a collector. A future collector must select typed edges by root role rather than treating every descriptor inside a closed bundle as perpetual execution input. -Only the selected successful closed attempt transfers into this push row. An older independent unknown attempt still retains its pin. Expiry alone never authorizes its deletion, a fresh command identity, or takeover. +Only eligible closed attempts transfer into the shared archive. An older denied initialization can retire independently of its successor after its active binding closes. An older independent unknown attempt still retains its pin. Expiry alone never authorizes its deletion, a fresh command identity, or takeover. ## Automatic service retirement @@ -44,10 +45,14 @@ Only the selected successful closed attempt transfers into this push row. An old Before each bounded keyset scan, the supervisor also recovers uncertain factory-owned terminal-release jobs from that same coordinator. This is necessary after a committed release removes its pin but loses its acknowledgement: a pin-only scan would no longer discover the still-charged command. Recovery retains the original SDK command, identity and receipt, including after coordinator close or scanner stop/restart. It does not retry compaction or replace a live producer's uncertain command. +An active denied initialization is deferred before release preparation or SDK admission. Successful repository startup retires its initial pin before exposing the route, and recovered positive startup discovers the original pin by the immutable initialization outcome’s exact incarnation/admission sequence. Pending startup retires an original known denial only after a successful Claim. Current maintenance fencing comes from the validated durable Cell Control and its live node advertisement; the receiver still independently checks its actual admitted fence and current Admin role. The existing tracked transition owns this constant-size work through cancellation. + Stopping the scanner requests stop between scans and joins its current work. It does not cancel an admitted release. Close and drain the publication coordinator separately. On owner succession, restart the service with current maintenance authority; original receipt lookup remains independent of that fresh authority. Production routing must use the SDK's ownership-aware transport rather than keeping a stopped owner's fixed local handle. ## Qualification and remaining work Native SHA-1/SHA-256 tests exercise quota release at the existing 4,096-pin cap, missing selected artifacts, current Admin and owner refusals, rollback at the final delete, original-command recovery after absence/lost acknowledgement/post-execution panic, scanner restart, automatic admission, uncertainty after pin disappearance, real SDK expiry, fresh-owner restore, immutable archives, removed original command bodies and current Read revocation. Existing multi-page policy tests retain their original page receipts after terminal archival. A separately prepared losing release is refused without replacing the winner's receipt. -These fixtures prove the covered transaction, receipt and lifecycle invariants. They do not establish throughput for 10,000 developers. Proof preparation is currently serialized by the repository scanner; node-wide fair verification admission, provider I/O budgets and full-history measurements remain mandatory. Production registration/startup/producer/reader conversion, initial staging uncertainty, retained-input Claim/adoption/repreparation, complete typed collection and isolated restore, file-backed intents/reports, OS containment, accelerated reads, physical rewriting and continuous hot-root maintenance remain open under the [implementation plan](../large-repository-implementation-plan.md) and [large-team requirements](../large-team-scalability.md). +Six typed initialization families additionally qualify both formats, immutable shared archives, denied-old/successful-new receipt separation, missing typed empty metadata, actual Admin/owner checks, last-write rollback, automatic release recovery after pin disappearance, SDK expiry, saved-body loss and fresh-owner restoration. The complete frozen-source publication suite passes 264 tests in 175.94 seconds with four threads and standard stacks. + +These fixtures prove the covered transaction, receipt and lifecycle invariants. They do not establish throughput for 10,000 developers. Proof preparation is currently serialized by the repository scanner; node-wide fair verification admission, provider I/O budgets and full-history measurements remain mandatory. Complete production producer/reader and background-service wiring, initial Begin/Claim/Renew/pre-registration uncertainty, retained-input Claim/adoption/repreparation, complete typed collection and isolated restore, file-backed intents/reports, OS containment, accelerated reads, physical rewriting and continuous hot-root maintenance remain open under the [implementation plan](../large-repository-implementation-plan.md) and [large-team requirements](../large-team-scalability.md). diff --git a/docs/evidence/serving-certified-browser-20261004.json b/docs/evidence/serving-certified-browser-20261004.json new file mode 100644 index 00000000..ab2baa7a --- /dev/null +++ b/docs/evidence/serving-certified-browser-20261004.json @@ -0,0 +1,253 @@ +{ + "source_files": 482, + "rust_files": 468, + "source_hash_digest": "380bd07d069a113df8abcb24464e9d0a7f16c1db86558fea9d4b3b6fc325d422", + "release_qualified": false, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 31.581, + "log": "/tmp/canopy-certified-browser-clippy.log", + "log_sha256": "1e0f4b4db116e6eda36a25a0af0901ab6934b363ca99f9db0c0ba886bd7b0e8f" + }, + { + "label": "focused-final", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "serving::", + "server::residency::tests::serving::", + "server::residency::tests::recovery::", + "catalog::native::tests", + "native_git::process::tests", + "git_cache::tests", + "git_objects::tests", + "git_read::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 120.624, + "log": "/tmp/canopy-certified-browser-focused-final.log", + "passed": 112, + "log_sha256": "cc61c30b6c2a3e49b8ccbff12c1fae43aa141c646664c306f56f355756f1cfb3" + }, + { + "label": "library", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked", + "--", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 311.84, + "log": "/tmp/canopy-certified-browser-library.log", + "passed": 690, + "failed": 5, + "nested_summaries_excluded": 2, + "known_failures": [ + "git_gateway::fetch::tests::reachability_stops_at_live_refs_without_scanning_other_history", + "object_reads::tests::byte_limited_page_advances_only_over_the_selected_prefix", + "object_reads::tests::duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts", + "object_reads::tests::insertion_cursor_finds_lower_oids_and_excludes_later_publications", + "object_reads::tests::small_increment_uses_bounded_sql_work_after_large_history" + ], + "publication_passed": 387, + "log_sha256": "bcd476cb8be246ff510b91076d41426485504571c6f31670b7ac90f76bdf0fd4" + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 45.627, + "log": "/tmp/canopy-certified-browser-workspace.log", + "passed": 3, + "log_sha256": "c38692f1ad948ed734afd1c87f79e1ec31c9655fece366de402df580a8a8cc4e" + }, + { + "label": "lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.359, + "log": "/tmp/canopy-certified-browser-lifecycle.log", + "passed": 2, + "log_sha256": "85dcf71fbe3e23af617affd0ca9822434e0198c6ddf77e2d7c2732516e6dcf28" + }, + { + "label": "drain", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.693, + "log": "/tmp/canopy-certified-browser-drain.log", + "passed": 4, + "log_sha256": "5b19b8343b428cccf7772157c4c9a581b8763f828d1ee6f124ec6c03b81085d1" + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 42.322, + "log": "/tmp/canopy-certified-browser-build.log", + "log_sha256": "9f43a14a9559bb08a537ffaf92b8d16a41d9d0141e9a4142233eaf63d1895798" + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 2.521, + "log": "/tmp/canopy-certified-browser-fmt.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.058, + "log": "/tmp/canopy-certified-browser-diff.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 42.561, + "log": "/tmp/canopy-certified-browser-harness.log", + "log_sha256": "a7176485db9a703c0bd8667e57658153b91ea6ad066df99f18ad012df27541e3" + } + ], + "execution_complete": true, + "unique_workspace_cases": 704, + "unique_workspace_passed": 699, + "unique_workspace_failed": 5, + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "sdk_manifest_pins": 5, + "sdk_lock_pins": 6, + "protected_hashes": { + "/Users/haipingfu/Github/canopy/.git/worktrees/canopy5/index": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "/Users/haipingfu/.codex/worktrees/packed-catalog-publication-pr/canopy/docs/archive/pr20-progress-through-8bb0ee7.md": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" + }, + "new_families": 6, + "harness_cases": 96, + "platform": "macOS / Rust 1.98.0; Linux-only fork cases not executed locally", + "scope": "Certified local browser and comparison object/ancestry consumers and bounded typed edge pages. Native fixtures physically verify pack sources, then install joint catalog facts and editorial pull metadata through trusted SQL. Not live producer/publication, remote routing, cold latency, streaming or full capacity qualification.", + "validated_source_parent": "b1bd813a50b8b93b833d4006b20c0b90713470a2", + "fetched_main": "9438bb865959fb975d5349ba8b9908b461653821", + "draft_diagnostics": [ + { + "log": "/tmp/canopy-certified-browser-check.log", + "exit_code": 101, + "reason": "Prototype compile diagnostics in constant export and test fixture OID parsing; corrected before final frozen-source qualification.", + "log_sha256": "c00694af162517d6d09b87b55831725694838dd0d1522d7344d6dffc7cb92e07" + }, + { + "log": "/tmp/canopy-certified-browser-focused-draft.log", + "exit_code": 101, + "reason": "Prototype compile diagnostics in constant export and test fixture OID parsing; corrected before final frozen-source qualification.", + "log_sha256": "f120600673a343d08dd9aaeb61d2415c1ffd78a99a31e4df3a1a23dea3bf08b8" + } + ], + "prior_ci": { + "head": "b1bd813a50b8b93b833d4006b20c0b90713470a2", + "runs": [ + 37229870509, + 37229867975 + ], + "rust": "Both actual failed logs contain exactly five legacy objects reader failures; 664 server cases pass in each.", + "harness": "both pass", + "not_qualification_for_new_source": true, + "primary_log_sha256": "0b50b178b53e857b4a413b1bd12f1349bda30b7c1541d7f268455bada8ca914a", + "secondary_log_sha256": "942955d03ba3f47a316cc4ec0e33200b42f2e858dc56c0a63c31225a1f213117" + }, + "checked_local_documentation_links": 96 +} diff --git a/docs/evidence/serving-certified-workspaces-20261004.json b/docs/evidence/serving-certified-workspaces-20261004.json new file mode 100644 index 00000000..a958df9a --- /dev/null +++ b/docs/evidence/serving-certified-workspaces-20261004.json @@ -0,0 +1,259 @@ +{ + "source_files": 487, + "rust_files": 473, + "source_hash_digest": "e618d9b235bdb2ad6b67c0168b8c6a65cf824c23196c54fb1aab0ad0d4ee1063", + "release_qualified": false, + "execution_complete": true, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 37.712, + "log": "/tmp/canopy-certified-workspace-clippy.log", + "log_sha256": "07eb3a24ed1f02c718862c62e27198d6464d8d5c009ccd20d84d0ee18a91ac00" + }, + { + "label": "focused-final", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "graph_spool::", + "serving::workspace::", + "certified_http_clone", + "native_receive_root_dispatch_preserves_known", + "final_current_policies_use", + "closed_owner_keeps_borrowed_generation", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 107.869, + "log": "/tmp/canopy-certified-workspace-focused-final.log", + "log_sha256": "420662234ec1a0de1f0f0b624b3644b505ee80bac92e228f0f73335346bbd188", + "passed": 17 + }, + { + "label": "library", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked" + ], + "exit_code": 101, + "seconds": 249.575, + "log": "/tmp/canopy-certified-workspace-library.log", + "log_sha256": "14961c459967e3e27007b4b4e2d8d55a596171e4016ed333f0d01af59078af4f", + "passed": 704, + "failed": 4, + "known_failures": [ + "object_reads::tests::byte_limited_page_advances_only_over_the_selected_prefix", + "object_reads::tests::duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts", + "object_reads::tests::insertion_cursor_finds_lower_oids_and_excludes_later_publications", + "object_reads::tests::small_increment_uses_bounded_sql_work_after_large_history" + ], + "nested_summaries_excluded": 2, + "publication_passed": 396 + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 53.432, + "log": "/tmp/canopy-certified-workspace-workspace.log", + "log_sha256": "73b6822d72f4082f55f1e7e9a41b391505acdcd32135bd96cade3328ffb7764e", + "passed": 3 + }, + { + "label": "lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.329, + "log": "/tmp/canopy-certified-workspace-lifecycle.log", + "log_sha256": "0b4f00db1414c119cb201e5c2a0f6aaa128b669858235b5da46bea25dfce9116", + "passed": 2 + }, + { + "label": "drain", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.741, + "log": "/tmp/canopy-certified-workspace-drain.log", + "log_sha256": "bd7b880b7b6c18aa2c032b4a8e8a9ea49264c27be7f8d470e77d35eaf7115142", + "passed": 4 + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 45.288, + "log": "/tmp/canopy-certified-workspace-build.log", + "log_sha256": "7eb95615ce99baf93cad6677aa35eee0068d1b026c92548557f1e9bf85c41e08" + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 2.254, + "log": "/tmp/canopy-certified-workspace-fmt.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.175, + "log": "/tmp/canopy-certified-workspace-diff.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 42.079, + "log": "/tmp/canopy-certified-workspace-harness.log", + "log_sha256": "5c8dac6c3a6376c2b124e1f6299548a0bed36a10123d54d3d403631f18b8633f" + } + ], + "unique_workspace_cases": 717, + "unique_workspace_passed": 713, + "unique_workspace_failed": 4, + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "sdk_manifest_pins": 5, + "sdk_lock_pins": 6, + "protected_hashes": { + "/Users/haipingfu/Github/canopy/.git/worktrees/canopy5/index": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "/Users/haipingfu/.codex/worktrees/packed-catalog-publication-pr/canopy/docs/archive/pr20-progress-through-8bb0ee7.md": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" + }, + "new_families": 13, + "relocated_cases": 1, + "harness_cases": 96, + "platform": "macOS / Rust 1.98.0; Linux-only fork cases not executed locally", + "scope": "Owned complete native forward workspaces and local HTTP/SSH fetch consumers. Real HTTP clones cover both formats; 10,000 ref names and a 532-parent merge qualify bounded traversal. Physical native verification plus trusted joint-fact installation isolates consumer semantics; not live write production, remote routing, large-history cold latency, workspace coalescing or large-team capacity.", + "validated_source_parent": "cc4a963c750d03c22483d6a9f12099a525114385", + "fetched_main": "9438bb865959fb975d5349ba8b9908b461653821", + "draft_diagnostics": [ + { + "source_hash_digest": "29c7cf4dcf3a11e828bcdbccbfee3798be2fad6e745ba8b94618d16fe787caff", + "exit_code": 101, + "server_passed": 683, + "server_failed": 5, + "reason": "Four legacy object readers and new short-lease workspace renewal case failed under broad parallel load. Metadata warmup moved before the short lease; returned workspace lifetime and completed construction remain asserted; ordinary body/membership reads have separate deadline tests.", + "log_sha256": "94125f6d417dc0c0f6d9e326af7d1f7ea38c8e49d2a3b5071c6ea0ea8c60e723" + }, + { + "source_hash_digest": "9618001776be59674c040b4f5a7df9b9ab4759043d8bf6a2bd286dbbaff40f4b", + "exit_code": 101, + "server_passed": 681, + "server_failed": 7, + "reason": "Initial full library run exposed three existing lease-setup/renewal timing assumptions beyond four legacy readers. Actual owned fixture renewal, controlled real-clock expiry injection at recovery, and event-based renewal assertions replace those assumptions.", + "log_sha256": "987a5aa5410a0a5bac22ac825bd53abb0b48d3566a6a72b784e156b3d6477a91" + } + ], + "isolated_timing_diagnostic": { + "passed": 3, + "exit_code": 0, + "log_sha256": "b183db4c52e2c7e051fea2837594765f115e580754e079dbaf0e809c693888cd", + "not_replacement_for_failed_broad_run": true + }, + "prior_ci": { + "head": "cc4a963c750d03c22483d6a9f12099a525114385", + "runs": [ + 37232379168, + 37232376754 + ], + "rust": "Both actual failed logs contain exactly five legacy objects reader failures; 670 server cases pass in each.", + "harness": "both pass", + "not_qualification_for_new_source": true, + "primary_log_sha256": "0d45c1f7145930e379d62f4c82ca68f9b7c4d067f122f2221e5507abb90005e9", + "secondary_log_sha256": "335496d75b03b7e0efae9ba7221a8a9dfcaffcab505d95854634dd23b2a242c3" + }, + "checked_local_documentation_links": 97 +} diff --git a/docs/evidence/serving-native-bodies-20261004.json b/docs/evidence/serving-native-bodies-20261004.json new file mode 100644 index 00000000..e3a20cb2 --- /dev/null +++ b/docs/evidence/serving-native-bodies-20261004.json @@ -0,0 +1,247 @@ +{ + "source_files": 478, + "rust_files": 464, + "source_hash_digest": "12d0ec70f91b2c04ee503e559ae6873e3081d78203dca33419a80bc40280eb90", + "release_qualified": false, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "log": "/tmp/canopy-native-body-clippy.log", + "preceding_unchanged_source": true, + "log_sha256": "fa8bed2a74b1f40588bc0aca7c274ce9ab106db852e8db873ced48304c0fc86c" + }, + { + "label": "focused-final", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "serving::", + "server::residency::tests::serving::", + "server::residency::tests::recovery::", + "catalog::native::tests", + "native_git::process::tests", + "git_cache::tests", + "git_objects::tests", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 112.04, + "log": "/tmp/canopy-native-body-focused-final.log", + "passed": 97, + "log_sha256": "afc8a0e412e9ce2c2da7c04a7b869c67e7a7dd7051d3282dbaf0d4dd2df9854f" + }, + { + "label": "library", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked", + "--", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 371.415, + "log": "/tmp/canopy-native-body-library.log", + "passed": 684, + "failed": 5, + "nested_summaries_excluded": 2, + "known_failures": [ + "git_gateway::fetch::tests::reachability_stops_at_live_refs_without_scanning_other_history", + "object_reads::tests::byte_limited_page_advances_only_over_the_selected_prefix", + "object_reads::tests::duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts", + "object_reads::tests::insertion_cursor_finds_lower_oids_and_excludes_later_publications", + "object_reads::tests::small_increment_uses_bounded_sql_work_after_large_history" + ], + "publication_passed": 385, + "log_sha256": "03748364c943d23c05df44bf766c58c5beff4830de63b44d73d4898b30b08a24" + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 100.294, + "log": "/tmp/canopy-native-body-workspace.log", + "passed": 3, + "log_sha256": "8297fe2608c19658b122fb4db86da24931f2a71902083d9efd4fbe0e514a737a" + }, + { + "label": "lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.681, + "log": "/tmp/canopy-native-body-lifecycle.log", + "passed": 2, + "log_sha256": "5e25685141fa6a95aaa6c1cbf330ac34c2ca007876d676951168b6ff71018e10" + }, + { + "label": "drain", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.864, + "log": "/tmp/canopy-native-body-drain.log", + "passed": 4, + "log_sha256": "68263cb86ae10ea23f325c9f1917d7839e77bd030c1eeeebfd850fffb5b01e44" + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 37.246, + "log": "/tmp/canopy-native-body-build.log", + "log_sha256": "7f94f507a00c1784dc00f3407768cb4a833bf6a184c4881c2ba0d9af2dd1ac1b" + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 1.216, + "log": "/tmp/canopy-native-body-fmt.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.056, + "log": "/tmp/canopy-native-body-diff.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 41.344, + "log": "/tmp/canopy-native-body-harness.log", + "log_sha256": "4c08a0a1ba770d431c13dcbcb9fc9f2578e0efea3754bc6d618fdc23c2015f5c" + } + ], + "execution_complete": true, + "unique_workspace_cases": 698, + "unique_workspace_passed": 693, + "unique_workspace_failed": 5, + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "sdk_manifest_pins": 5, + "sdk_lock_pins": 6, + "protected_hashes": { + "/Users/haipingfu/Github/canopy/.git/worktrees/canopy5/index": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "/Users/haipingfu/.codex/worktrees/packed-catalog-publication-pr/canopy/docs/archive/pr20-progress-through-8bb0ee7.md": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" + }, + "new_families": 10, + "harness_cases": 96, + "platform": "macOS / Rust 1.98.0; Linux-only fork cases not executed locally", + "scope": "Certified bounded native body API and resident pack reuse; callers beyond refs and remote routes remain to be converted. Native serving fixtures use trusted catalog installation; not producer/publication, cold HTTP latency or full capacity qualification.", + "validated_source_parent": "b89d92980fbc9ae515b4b5376ec93415f0f88d68", + "fetched_main": "9438bb865959fb975d5349ba8b9908b461653821", + "draft_diagnostics": [ + { + "log": "/tmp/canopy-native-body-clippy-draft.log", + "exit_code": 101, + "reason": "Clippy byte_char_slices rejected a test-only newline iterator; corrected before frozen qualification.", + "log_sha256": "db1863bb164d3e296d34a171b1af6a10a094b5f36de57e461bb676fd300c1ee0" + } + ], + "clippy_seconds": 38.39, + "prior_ci": { + "head": "b89d92980fbc9ae515b4b5376ec93415f0f88d68", + "runs": [ + 37227358897, + 37227355827 + ], + "rust": "Both actual failed logs contain exactly five legacy objects reader failures; 654 server cases pass in each.", + "harness": "both pass", + "not_qualification_for_new_source": true, + "primary_log_sha256": "4dd14218fc7a8d57019e54782826794ea29a05ea96441d636a0acc3938d2e135", + "secondary_log_sha256": "0620c90b05f39addfd7589feab437c731387047107b250a30a3d4b103a8b5828" + }, + "checked_local_documentation_links": 92 +} diff --git a/docs/evidence/serving-native-write-base-20261004.json b/docs/evidence/serving-native-write-base-20261004.json new file mode 100644 index 00000000..1be22d04 --- /dev/null +++ b/docs/evidence/serving-native-write-base-20261004.json @@ -0,0 +1,357 @@ +{ + "checkpoint": "immutable source cursor and native write-base conversion", + "parent": "6ab78d679839fa8e5fc80002fa529ca61e799874", + "validation": { + "source_files": 487, + "rust_files": 473, + "source_hash_digest": "5be32a60d63d87754e87f1b2fd0112ad5c2223d85da480b74446f22733c0511e", + "release_qualified": false, + "execution_complete": true, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 36.465, + "log": "/tmp/canopy-source-cursor-clippy-final.log", + "log_sha256": "a3e34c494a21d34d1eef4ea02466734e69238fe85b33491412b29da6b03805db", + "summaries": [], + "failed_cases": [] + }, + { + "label": "focused", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "sources::tests::changes::", + "native_write_base", + "native_refs_tests", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 114.256, + "log": "/tmp/canopy-source-cursor-focused-final.log", + "log_sha256": "2da7defefb14a3d7c43569684ab19689da2d175f0e604d30e76a1058554070d5", + "summaries": [ + [ + 9, + 0, + 0, + 0, + 684 + ] + ], + "failed_cases": [] + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--locked" + ], + "exit_code": 101, + "seconds": 326.624, + "log": "/tmp/canopy-source-cursor-workspace-final.log", + "log_sha256": "538d0ddf6ccb4121a02fd4050d5861f2bf6967f57281c04805eefbab8158ab3b", + "summaries": [ + [ + 6, + 0, + 0, + 0, + 0 + ], + [ + 14, + 0, + 0, + 0, + 0 + ], + [ + 1, + 0, + 0, + 0, + 692 + ], + [ + 1, + 0, + 0, + 0, + 692 + ], + [ + 693, + 0, + 0, + 0, + 0 + ], + [ + 2, + 0, + 0, + 0, + 0 + ], + [ + 12, + 1, + 0, + 0, + 0 + ] + ], + "failed_cases": [ + "directory_reservations_recover_two_distinct_repository_cells" + ] + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 45.072, + "log": "/tmp/canopy-source-cursor-build-final.log", + "log_sha256": "07d38a1434fc1584b81ac2432db3cf79c5f089b92432961cfbb7ac57f5421f98", + "summaries": [], + "failed_cases": [] + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 1.262, + "log": "/tmp/canopy-source-cursor-fmt-final.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "summaries": [], + "failed_cases": [] + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.063, + "log": "/tmp/canopy-source-cursor-diff-final.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "summaries": [], + "failed_cases": [] + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 50.538, + "log": "/tmp/canopy-source-cursor-harness-final.log", + "log_sha256": "df4c5131cbc670755c8370851cdc37d11d296cf66378163bdb944d790dda7bdd", + "summaries": [], + "failed_cases": [] + }, + { + "label": "selected_lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "log": "/tmp/canopy-source-cursor-lifecycle-final.log", + "log_sha256": "db9c91ccc85d41f0c15282b6f66309c68ec6f5e0540a303f5ce87e7158b28bfd", + "passed": 9, + "test_seconds": 6.64 + } + ], + "unique_library_cases": 713, + "library_passed": 713, + "library_failed": 0, + "server_library_passed": 693, + "publication_library_passed": 398, + "binary_passed": 2, + "directory_passed": 12, + "directory_failed": 1, + "selected_lifecycle_passed": 9, + "unique_executed_rust_cases": 737, + "unique_executed_rust_passed": 736, + "unique_executed_rust_failed": 1, + "nested_and_focused_runs_excluded": true, + "full_workspace_halted_at": "directory_cell", + "unrun_after_halt": [ + "multi_server full suite", + "owner_restart", + "repository_cell", + "smart_http", + "doc tests" + ], + "provider_qualification_run": false, + "linux_only_cases_run": false, + "source_unchanged": true + }, + "static": { + "source_digest": "5be32a60d63d87754e87f1b2fd0112ad5c2223d85da480b74446f22733c0511e", + "source_unchanged": true, + "original_index_sha256": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "archive_sha256": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e", + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "dependency_declarations": 5, + "locked_sources": 6 + }, + "preceding_actual_ci": [ + { + "head": "6ab78d679839fa8e5fc80002fa529ca61e799874", + "run": 37238249979, + "job": 111541595278, + "conclusion": "FAILURE", + "server_passed": 684, + "server_failed": 4, + "log": "/tmp/canopy-pr34-6ab-failed.log", + "sha256": "15297c8d83aef2f27c756463d628fb67a80c3450ae8a2e51fb8783ea8ce2b048" + }, + { + "head": "6ab78d679839fa8e5fc80002fa529ca61e799874", + "run": 37238247841, + "job": 111541588793, + "conclusion": "FAILURE", + "server_passed": 684, + "server_failed": 4, + "log": "/tmp/canopy-pr34-6ab-secondary-failed.log", + "sha256": "704efd84fb15b9ce9ab15dfa1c910dc3a6f233aaad3088227e67751e509506af" + } + ], + "failed_draft_diagnostics": { + "source_files": 487, + "rust_files": 473, + "source_hash_digest": "ccc988e21dfdb7a4b00a49fcd661afa5a26efab74228b0c0ed9b99881a37ef9c", + "release_qualified": false, + "execution_complete": false, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 48.598, + "log": "/tmp/canopy-source-cursor-clippy-final.log", + "log_sha256": "f9d77bb384c10451e2713eb5be727738229fd24499c1cba23ffe234ccc090fc8", + "summaries": [], + "failed_cases": [], + "historical_log_path_reused": true, + "digest_only_retained": true + }, + { + "label": "focused", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "sources::tests::changes::", + "native_write_base", + "native_refs_tests", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 89.736, + "log": "/tmp/canopy-source-cursor-focused-before-ref-warm.log", + "log_sha256": "6a513f8d80ab1d590e513dca0b8ed2c74a8bc94b19ddd1323e000ce395a4e4a7", + "summaries": [ + [ + 8, + 1, + 0, + 0, + 684 + ] + ], + "failed_cases": [ + "packs::publication::tests::serving::workspace::native_write_base_cancellation_and_revocation_retain_real_provider_work_until_drain" + ] + } + ], + "log_note": "failed focused log is preserved at /tmp/canopy-source-cursor-focused-before-ref-warm.log; original final-log path was reused only after source repair", + "preserved_focused_log_sha256": "6a513f8d80ab1d590e513dca0b8ed2c74a8bc94b19ddd1323e000ce395a4e4a7" + }, + "scope_limits": [ + "Owned HTTP/SSH/generated write publication remains unconverted", + "Cold native baseline still downloads and verifies the selected catalog per request", + "Difference cursor supports root-pair increments but production cold constructor starts at None", + "Physical native fixtures qualify consumers, not live producer publication", + "Full integration/provider and capacity gates remain open", + "No compatibility table, command fallback or skipped test added" + ] +} diff --git a/docs/evidence/serving-owner-20261004.json b/docs/evidence/serving-owner-20261004.json new file mode 100644 index 00000000..5a87c8d8 --- /dev/null +++ b/docs/evidence/serving-owner-20261004.json @@ -0,0 +1,212 @@ +{ + "source_files": 468, + "rust_files": 454, + "source_hash_digest": "15b8b304b6e94849fbc3ebc156f8004efb4b92aaf08faedd6f2a9e634d52c732", + "phases": [ + { + "label": "clippy", + "exit_code": 0, + "log": "/tmp/canopy-serving-owner-clippy.log", + "preceding_unchanged_source": true, + "log_sha256": "4bee68cce0c5bc16353ee4eade3198820acc007be132601c6f1aaac76052d1e4" + }, + { + "label": "focused", + "exit_code": 0, + "log": "/tmp/canopy-serving-owner-focused.log", + "preceding_unchanged_source": true, + "log_sha256": "049a023df5e6a7f558f229dccefbc0bb562e6403c65fdd81f336a6d3522bfcbc" + }, + { + "label": "library", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked", + "--", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 306.149, + "log": "/tmp/canopy-serving-owner-library.log", + "passed": 659, + "failed": 5, + "ignored": 0, + "publication_passed": 370, + "known_failures": [ + "git_gateway::fetch::tests::reachability_stops_at_live_refs_without_scanning_other_history", + "object_reads::tests::byte_limited_page_advances_only_over_the_selected_prefix", + "object_reads::tests::duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts", + "object_reads::tests::insertion_cursor_finds_lower_oids_and_excludes_later_publications", + "object_reads::tests::small_increment_uses_bounded_sql_work_after_large_history" + ], + "nested_summaries_excluded": 2, + "log_sha256": "0270e2739cb943f4015ad93caecf7e6d1d592e794409e3aedff735302650aab9" + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 55.346, + "log": "/tmp/canopy-serving-owner-workspace.log", + "passed": 3, + "log_sha256": "e624ecbce6f17a4c7b0b3596c91923fdde6d409653529ea5e6182a743e1c547c" + }, + { + "label": "lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 2.578, + "log": "/tmp/canopy-serving-owner-lifecycle.log", + "passed": 2, + "log_sha256": "144e78806ba656a04f0d78e90d743461f1abb0fe3391673f5c5c8e969553fee5" + }, + { + "label": "drain", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 2.542, + "log": "/tmp/canopy-serving-owner-drain.log", + "passed": 4, + "log_sha256": "6a024d55ca170f22153fd32e74cfb309b29def579a59646ff551ecef2312212b" + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 37.259, + "log": "/tmp/canopy-serving-owner-build.log", + "log_sha256": "60125a374783a064cfd21cb84bb2e3bdba615a41d566065c2e528186072e25ea" + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 1.148, + "log": "/tmp/canopy-serving-owner-fmt.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.048, + "log": "/tmp/canopy-serving-owner-diff.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 41.109, + "log": "/tmp/canopy-serving-owner-harness.log", + "log_sha256": "1d58c5218a1e090b434b641ebbcda2d16f500a7f37bdcaae710d64d00ce43e53" + } + ], + "release_qualified": false, + "driver_diagnostic": "The driver exited 1 twice after completed test commands: the startup count initially used the lifecycle prefix rather than catalog_initialization; the initial integration selection expected Linux-only fork tests on macOS. Actual logs and exit codes are retained. Corrected assertions reuse completed commands and add four unexecuted portable lifecycle cases; no successful command is repeated.", + "unique_workspace_cases": 673, + "unique_workspace_passed": 668, + "unique_workspace_failed": 5, + "execution_complete": true, + "clippy_seconds": 29.19, + "harness_cases": 96, + "scope": "Workspace library plus nine selected portable multi_server cases; focused lifecycle cases are included in the library total. This is not full production or capacity qualification.", + "platform": "macOS / Rust 1.98.0; Linux-only fork cases not executed locally", + "environment": { + "CARGO_INCREMENTAL": "0", + "CARGO_TARGET_DIR": "shared ignored build directory; no deployment/demo restart" + }, + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "sdk_manifest_pins": 5, + "sdk_lock_pins": 6, + "protected_hashes": { + "/Users/haipingfu/Github/canopy/.git/worktrees/canopy5/index": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "/Users/haipingfu/.codex/worktrees/packed-catalog-publication-pr/canopy/docs/archive/pr20-progress-through-8bb0ee7.md": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" + }, + "validated_source_parent": "2012f867cb5ab50d460cb67c942a6fc518f49c36", + "fetched_main": "9438bb865959fb975d5349ba8b9908b461653821", + "prior_ci": { + "head": "2012f867cb5ab50d460cb67c942a6fc518f49c36", + "rust": "failed: five legacy objects readers", + "harness": "passed", + "runs": [ + 37214232434, + 37214228893 + ], + "not_qualification_for_new_source": true + }, + "checked_local_documentation_links": 79 +} diff --git a/docs/evidence/serving-pool-20261004.json b/docs/evidence/serving-pool-20261004.json new file mode 100644 index 00000000..1188a101 --- /dev/null +++ b/docs/evidence/serving-pool-20261004.json @@ -0,0 +1,219 @@ +{ + "source_files": 471, + "rust_files": 457, + "source_hash_digest": "a82809971ed44451e9be3c0fbf7d088d6206965773d63c8507c6a1d758b3adc5", + "phases": [ + { + "label": "clippy", + "exit_code": 0, + "log": "/tmp/canopy-serving-pool-clippy.log", + "preceding_unchanged_source": true, + "log_sha256": "a5f2abc6d3c813f9db7d6c40de6bd827a277bad6c99d683ec5ca1b04d65837d2" + }, + { + "label": "focused", + "exit_code": 0, + "log": "/tmp/canopy-serving-pool-focused.log", + "preceding_unchanged_source": true, + "log_sha256": "e9437cc92d829b7c6f0c6f9522014044f11d00047a690d1a119045174b9d05f7" + }, + { + "label": "library", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked", + "--", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 301.816, + "log": "/tmp/canopy-serving-pool-library.log", + "passed": 667, + "failed": 5, + "ignored": 0, + "publication_passed": 376, + "known_failures": [ + "git_gateway::fetch::tests::reachability_stops_at_live_refs_without_scanning_other_history", + "object_reads::tests::byte_limited_page_advances_only_over_the_selected_prefix", + "object_reads::tests::duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts", + "object_reads::tests::insertion_cursor_finds_lower_oids_and_excludes_later_publications", + "object_reads::tests::small_increment_uses_bounded_sql_work_after_large_history" + ], + "nested_summaries_excluded": 2, + "log_sha256": "4f580b7c5c2fbd85367d45e474c460950c37d122eb2814cc95b80a70ce17f400" + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 48.787, + "log": "/tmp/canopy-serving-pool-workspace.log", + "passed": 3, + "log_sha256": "7ae4766515dd848464e1753c7ae309b2484062a01cabf2ad5002934dc3e0fb18" + }, + { + "label": "lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.451, + "log": "/tmp/canopy-serving-pool-lifecycle.log", + "passed": 2, + "log_sha256": "835770997bec08bd451b5f9e49afa90219501017dcfb87a6806154c7e109a658" + }, + { + "label": "drain", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.756, + "log": "/tmp/canopy-serving-pool-drain.log", + "passed": 4, + "log_sha256": "07a1f60ef4ea65014dabc54333f7daf0b76f15c9257ec9614a95f4826d026ddf" + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 32.158, + "log": "/tmp/canopy-serving-pool-build.log", + "log_sha256": "86cfe0a814d94b2df2b249de7096d48b8cda847434f64c17ee0cac482337b190" + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 1.352, + "log": "/tmp/canopy-serving-pool-fmt.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.054, + "log": "/tmp/canopy-serving-pool-diff.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 41.818, + "log": "/tmp/canopy-serving-pool-harness.log", + "log_sha256": "c82b12af80f8b7db1b4bc7a78bb324409155b13661bc503dae43f9dfe5c305ff" + } + ], + "release_qualified": false, + "unique_workspace_cases": 681, + "unique_workspace_passed": 676, + "unique_workspace_failed": 5, + "execution_complete": true, + "clippy_seconds": 28.89, + "harness_cases": 96, + "focused_cases": 55, + "new_pool_families": 6, + "new_production_manager_families": 2, + "draft_diagnostic": { + "log": "/tmp/canopy-serving-pool-check-draft.log", + "exit_code": 101, + "reason": "Arc was not imported in lib.rs; fixed with fully qualified std::sync::Arc before the final frozen-source tests." + }, + "scope": "Workspace library plus nine selected portable multi_server cases; focused lifecycle cases are included in the library total. This is not full production or capacity qualification.", + "platform": "macOS / Rust 1.98.0; Linux-only fork cases not executed locally", + "environment": { + "CARGO_INCREMENTAL": "0", + "CARGO_TARGET_DIR": "shared ignored build directory; no deployment/demo restart" + }, + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "sdk_manifest_pins": 5, + "sdk_lock_pins": 6, + "protected_hashes": { + "/Users/haipingfu/Github/canopy/.git/worktrees/canopy5/index": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "/Users/haipingfu/.codex/worktrees/packed-catalog-publication-pr/canopy/docs/archive/pr20-progress-through-8bb0ee7.md": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" + }, + "validated_source_parent": "b21f9d7c74b85eab945146df1c4778433a350a72", + "fetched_main": "9438bb865959fb975d5349ba8b9908b461653821", + "prior_ci": { + "head": "b21f9d7c74b85eab945146df1c4778433a350a72", + "rust": "failed: five legacy objects readers", + "harness": "passed", + "runs": [ + 37219758565, + 37219757347 + ], + "not_qualification_for_new_source": true + }, + "checked_local_documentation_links": 86 +} diff --git a/docs/evidence/serving-pool-shutdown-20261004.json b/docs/evidence/serving-pool-shutdown-20261004.json new file mode 100644 index 00000000..cd06dbbb --- /dev/null +++ b/docs/evidence/serving-pool-shutdown-20261004.json @@ -0,0 +1,247 @@ +{ + "source_files": 471, + "rust_files": 457, + "source_hash_digest": "c88bdd413d5308745295cffa77ef035e6f8b1b777a61e368561a4ef0cf32931b", + "phases": [ + { + "label": "clippy", + "exit_code": 0, + "log": "/tmp/canopy-serving-pool-shutdown-clippy.log", + "preceding_unchanged_source": true, + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "log_sha256": "5ef8cbb8034e3f6032c9161794579ed19d5c168b42c47fc41fb7758a3032fe44" + }, + { + "label": "focused", + "exit_code": 0, + "log": "/tmp/canopy-serving-pool-shutdown-focused.log", + "preceding_unchanged_source": true, + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "packs::publication::tests::serving", + "server::residency::tests::serving", + "server::residency::tests::recovery", + "--test-threads=4" + ], + "log_sha256": "6f7daac8dec2014df2651039a338031bb58a91ffd91e6ce012cdbb30732de744" + }, + { + "label": "library", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked", + "--", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 297.396, + "log": "/tmp/canopy-serving-pool-shutdown-library.log", + "passed": 668, + "failed": 5, + "ignored": 0, + "publication_passed": 376, + "known_failures": [ + "git_gateway::fetch::tests::reachability_stops_at_live_refs_without_scanning_other_history", + "object_reads::tests::byte_limited_page_advances_only_over_the_selected_prefix", + "object_reads::tests::duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts", + "object_reads::tests::insertion_cursor_finds_lower_oids_and_excludes_later_publications", + "object_reads::tests::small_increment_uses_bounded_sql_work_after_large_history" + ], + "nested_summaries_excluded": 2, + "log_sha256": "f2b4e260647098a519396f512e5188864625a7e120bcbd95351bd5e69c4afebe" + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 50.397, + "log": "/tmp/canopy-serving-pool-shutdown-workspace.log", + "passed": 3, + "log_sha256": "eb093ab093b3b29be1522aaf19f3ac0e782e5cc70fef1407ef994fa08edf27f9" + }, + { + "label": "lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.486, + "log": "/tmp/canopy-serving-pool-shutdown-lifecycle.log", + "passed": 2, + "log_sha256": "7eb081b7ba4a3aeb5f29be975711c815618b811c3b84a4ac8979047470284924" + }, + { + "label": "drain", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.962, + "log": "/tmp/canopy-serving-pool-shutdown-drain.log", + "passed": 4, + "log_sha256": "0a78c2c31b80bdefe5b53fdbb15cfda7a888169b87df5a4964874dfd666d5883" + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 34.583, + "log": "/tmp/canopy-serving-pool-shutdown-build.log", + "log_sha256": "57745bcaf7d13b8145fc270525fdf83599e83b9a56b39ac82cab281f54829a32" + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 1.164, + "log": "/tmp/canopy-serving-pool-shutdown-fmt.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.049, + "log": "/tmp/canopy-serving-pool-shutdown-diff.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 40.636, + "log": "/tmp/canopy-serving-pool-shutdown-harness.log", + "log_sha256": "0d7c467046352f9152f9998f1140b9221739e6a684843665020c2e8ea3081ef8" + } + ], + "release_qualified": false, + "unique_workspace_cases": 682, + "unique_workspace_passed": 677, + "unique_workspace_failed": 5, + "execution_complete": true, + "clippy_seconds": 31.41, + "harness_cases": 96, + "focused_cases": 56, + "focused_seconds": 19.4, + "pool_families": 6, + "production_manager_families": 3, + "new_production_manager_families": 1, + "regression_before_fix": { + "log": "/tmp/canopy-serving-pool-shutdown-regression-red.log", + "log_sha256": "4cf72fdd260bd4f4db57e9fa2e87063fe0d59648c5c7ea4066fafb019c736788", + "exit_code": 101, + "reason": "An unpublished resident constructor exposed the serving capability before its lifecycle owner was registered in shutdown inventory." + }, + "scope": "Workspace library plus nine selected portable multi_server cases; focused lifecycle cases are included in the library total. This is not full production or capacity qualification.", + "platform": "macOS / Rust 1.98.0; Linux-only fork cases not executed locally", + "environment": { + "CARGO_INCREMENTAL": "0", + "CARGO_TARGET_DIR": "shared ignored build directory; no deployment/demo restart" + }, + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "sdk_manifest_pins": 5, + "sdk_lock_pins": 6, + "protected_hashes": { + "/Users/haipingfu/Github/canopy/.git/worktrees/canopy5/index": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "/Users/haipingfu/.codex/worktrees/packed-catalog-publication-pr/canopy/docs/archive/pr20-progress-through-8bb0ee7.md": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" + }, + "validated_source_parent": "02307fd3b7de774ed47943548d460ddaa329911b", + "fetched_main": "9438bb865959fb975d5349ba8b9908b461653821", + "prior_ci": { + "head": "02307fd3b7de774ed47943548d460ddaa329911b", + "rust": "failed: five legacy objects readers", + "harness": "passed", + "runs": [ + 37222526207, + 37222523633 + ], + "not_qualification_for_new_source": true + }, + "checked_local_documentation_links": 87 +} diff --git a/docs/evidence/serving-refs-20261004.json b/docs/evidence/serving-refs-20261004.json new file mode 100644 index 00000000..68e64807 --- /dev/null +++ b/docs/evidence/serving-refs-20261004.json @@ -0,0 +1,256 @@ +{ + "source_files": 474, + "rust_files": 460, + "source_hash_digest": "8f29f85e5224490ae647603bb51ebf64add25d4be3c1a748b559cf6418ce80f1", + "phases": [ + { + "label": "clippy", + "exit_code": 0, + "log": "/tmp/canopy-serving-refs-clippy.log", + "preceding_unchanged_source": true, + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "log_sha256": "2ef814fea8430b2d9875931a8da0eec6994f6db900678285b2abbca6b2610eb2" + }, + { + "label": "focused", + "exit_code": 0, + "log": "/tmp/canopy-serving-refs-focused.log", + "preceding_unchanged_source": true, + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "packs::publication::tests::serving", + "server::residency::tests::serving", + "server::residency::tests::recovery", + "--test-threads=4" + ], + "log_sha256": "38824abbd62d5575e20bd5da172f6a577484789398ee143e6329b5e5b764afd3" + }, + { + "label": "library", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked", + "--", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 319.161, + "log": "/tmp/canopy-serving-refs-library.log", + "passed": 674, + "failed": 5, + "ignored": 0, + "publication_passed": 381, + "known_failures": [ + "git_gateway::fetch::tests::reachability_stops_at_live_refs_without_scanning_other_history", + "object_reads::tests::byte_limited_page_advances_only_over_the_selected_prefix", + "object_reads::tests::duplicate_rollback_and_deletion_do_not_hide_subsequent_inserts", + "object_reads::tests::insertion_cursor_finds_lower_oids_and_excludes_later_publications", + "object_reads::tests::small_increment_uses_bounded_sql_work_after_large_history" + ], + "nested_summaries_excluded": 2, + "log_sha256": "fc0247c2f4bca0d77357470ff286b40886a169ab16f62bcdf6e18773098f5920" + }, + { + "label": "workspace", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "workspace::", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 49.394, + "log": "/tmp/canopy-serving-refs-workspace.log", + "passed": 3, + "log_sha256": "a337ef4df179f5a331a9e5a4b57933eaac2d5c0d695ab1cd1a6c86eece61b56d" + }, + { + "label": "lifecycle", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::fork::", + "lifecycle::cancelled_prebound_startup", + "lifecycle::runtime_destruction", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 2.4, + "log": "/tmp/canopy-serving-refs-lifecycle.log", + "passed": 2, + "log_sha256": "6acc35040ce2fe43604a266a74cb5872201853bf3f4b52b8a34871c2176c5052" + }, + { + "label": "drain", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "multi_server", + "--locked", + "--", + "lifecycle::cancelled_startup_keeps_workspace", + "lifecycle::dropped_handle_and_cancelled_shutdown", + "lifecycle::failed_drain_retains_workspace", + "lifecycle::startup_rejects_ignored_conditional", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 1.823, + "log": "/tmp/canopy-serving-refs-drain.log", + "passed": 4, + "log_sha256": "0a023f055205a8f1e78b323294adec111fe2706ad0b0b85ab279da45a058f220" + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 34.969, + "log": "/tmp/canopy-serving-refs-build.log", + "log_sha256": "b307219c288f74812291c45ea66b0e6286a48003bff3756a846d6ff8d429a545" + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 1.181, + "log": "/tmp/canopy-serving-refs-fmt.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.054, + "log": "/tmp/canopy-serving-refs-diff.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 41.981, + "log": "/tmp/canopy-serving-refs-harness.log", + "log_sha256": "f8961ead9be0a4008e6c4f7837f0e66f80ba739d0b42551ca7aef8c97b6e133c" + } + ], + "release_qualified": false, + "unique_workspace_cases": 688, + "unique_workspace_passed": 683, + "unique_workspace_failed": 5, + "execution_complete": true, + "clippy_seconds": 28.37, + "harness_cases": 96, + "focused_cases": 62, + "focused_seconds": 20.59, + "pool_families": 6, + "production_manager_families": 4, + "new_ref_families": 5, + "new_production_manager_families": 1, + "draft_diagnostics": [ + { + "log": "/tmp/canopy-serving-refs-initial-focused.log", + "exit_code": 101, + "log_sha256": "fb37089273f7a353780ce91979d4b2b38a8393d1a6aba4e4d010b1e6f79f6d9e", + "reason": "The new fixture supplied bare records to build_sorted, which requires Result records; fixed before qualification." + }, + { + "log": "/tmp/canopy-serving-refs-new-focused.log", + "exit_code": 101, + "log_sha256": "c31fa05cb0010c07c177811e8d6862f481648b17e1fd57d03ff2cd68d626c6ee", + "reason": "Three new fixture roots used invalid artifact namespaces, and the HTTP fixture supplied hex rather than a canonical UUID; corrected before qualification." + } + ], + "scope": "Local browser ref conversion and owned immutable ref reads. Workspace library plus nine selected portable multi_server cases; focused lifecycle cases are included in the library total. This is not full production or capacity qualification.", + "platform": "macOS / Rust 1.98.0; Linux-only fork cases not executed locally", + "environment": { + "CARGO_INCREMENTAL": "0", + "CARGO_TARGET_DIR": "shared ignored build directory; no deployment/demo restart" + }, + "sdk_revision": "161067f5a21703b3e257024bcb64e565fd9657b4", + "sdk_manifest_pins": 5, + "sdk_lock_pins": 6, + "protected_hashes": { + "/Users/haipingfu/Github/canopy/.git/worktrees/canopy5/index": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "/Users/haipingfu/.codex/worktrees/packed-catalog-publication-pr/canopy/docs/archive/pr20-progress-through-8bb0ee7.md": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" + }, + "validated_source_parent": "1e32347a411fe24245e34638221e79eb8968b241", + "fetched_main": "9438bb865959fb975d5349ba8b9908b461653821", + "prior_ci": { + "head": "1e32347a411fe24245e34638221e79eb8968b241", + "rust": "failed: five legacy objects readers", + "harness": "passed", + "runs": [ + 37224582653, + 37224580445 + ], + "not_qualification_for_new_source": true + }, + "checked_local_documentation_links": 90 +} diff --git a/docs/evidence/staging-physical-workers-20261004.json b/docs/evidence/staging-physical-workers-20261004.json new file mode 100644 index 00000000..8f1e0884 --- /dev/null +++ b/docs/evidence/staging-physical-workers-20261004.json @@ -0,0 +1,649 @@ +{ + "checkpoint": "staging physical worker ownership and publication-phase ceiling qualification", + "parent": "6101d5869f71dd836876266ba8d5aec0ac7965b0", + "validation": { + "source_files": 488, + "rust_files": 474, + "source_hash_digest": "c76cad71a122baa769df897444e87eb8d7ef4a3b8e46ecaedbb2a9e35ff833fa", + "release_qualified": false, + "execution_complete": true, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 32.272, + "log": "/tmp/canopy-stage-physical-clippy-final.log", + "log_sha256": "3df8a565e1aa6a6df71d04cf363c444c0bacaaf546f61fc85236f6da9013b45d", + "summaries": [], + "failed_cases": [] + }, + { + "label": "focused", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "physical_creating_worker", + "physical_bound_worker", + "git_http::capture::tests::", + "native_receive_stages_verifies_and_publishes_then_clones_after_cache_loss", + "native_receive_prepares_bounded_immutable_root_completion_from_registered_custody", + "bound_service_phase_handoff_and_residence_cap_fence_existing_bases_and_inflight_workers", + "closed_owner_keeps_borrowed_generation_renewing_until_last_snapshot_clone_drops", + "staging_service::publication::bound_final_ceiling", + "staging_service::publication::bound_final_queued_transport", + "staging_service::publication::bound_final_exact_recovery", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 86.42, + "log": "/tmp/canopy-stage-physical-focused-final.log", + "log_sha256": "75f1fbb56b460e9397fd77702b61a9bb1741172473478a80e339cc08e12ed239", + "summaries": [ + [ + 11, + 0, + 0, + 0, + 684 + ] + ], + "failed_cases": [] + }, + { + "label": "libraries", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked" + ], + "exit_code": 0, + "seconds": 271.387, + "log": "/tmp/canopy-stage-physical-libraries-final.log", + "log_sha256": "60b9957c10e2c28124efe01fb0e0c891d7fe7e3a9cad7e6f1daa5c2865da6001", + "summaries": [ + [ + 6, + 0, + 0, + 0, + 0 + ], + [ + 14, + 0, + 0, + 0, + 0 + ], + [ + 1, + 0, + 0, + 0, + 694 + ], + [ + 1, + 0, + 0, + 0, + 694 + ], + [ + 695, + 0, + 0, + 0, + 0 + ] + ], + "failed_cases": [] + }, + { + "label": "directory", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--test", + "directory_cell", + "--locked" + ], + "exit_code": 101, + "seconds": 53.858, + "log": "/tmp/canopy-stage-physical-directory-final.log", + "log_sha256": "d5101da1851c49576bdd6498999584dd657e2d2dfd9c8660320beb854df5ca09", + "summaries": [ + [ + 12, + 1, + 0, + 0, + 0 + ] + ], + "failed_cases": [ + "directory_reservations_recover_two_distinct_repository_cells" + ] + }, + { + "label": "binary", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--bin", + "canopy", + "--locked" + ], + "exit_code": 0, + "seconds": 6.919, + "log": "/tmp/canopy-stage-physical-binary-final.log", + "log_sha256": "8b2e1fa03ccf2587724ee53f19dfed77dc4d216c29d6bc9922b15bec6c25114d", + "summaries": [ + [ + 2, + 0, + 0, + 0, + 0 + ] + ], + "failed_cases": [] + }, + { + "label": "build", + "command": [ + "cargo", + "+1.98.0", + "build", + "--locked", + "--bin", + "canopy" + ], + "exit_code": 0, + "seconds": 62.313, + "log": "/tmp/canopy-stage-physical-build-final.log", + "log_sha256": "962e16493d5d1c437445aa2436052d9ac082faf97a33efdfa21519f783b70a06", + "summaries": [], + "failed_cases": [] + }, + { + "label": "fmt", + "command": [ + "cargo", + "+1.98.0", + "fmt", + "--all", + "--", + "--check" + ], + "exit_code": 0, + "seconds": 1.222, + "log": "/tmp/canopy-stage-physical-fmt-final.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "summaries": [], + "failed_cases": [] + }, + { + "label": "diff", + "command": [ + "git", + "diff", + "--check" + ], + "exit_code": 0, + "seconds": 0.066, + "log": "/tmp/canopy-stage-physical-diff-final.log", + "log_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "summaries": [], + "failed_cases": [] + }, + { + "label": "harness", + "command": [ + "python3", + "-B", + "-m", + "unittest", + "discover", + "-s", + "scripts", + "-p", + "test_*.py" + ], + "exit_code": 0, + "seconds": 41.298, + "log": "/tmp/canopy-stage-physical-harness-final.log", + "log_sha256": "87b82abdb6530000635d38f0cc5f7acb83a174027f6f59c207fa93ab1eb4ce2b", + "summaries": [], + "failed_cases": [] + } + ] + }, + "unique_rust_executed": 730, + "rust_passed": 729, + "rust_failed": 1, + "library_unique_passed": 715, + "server_library_unique_passed": 695, + "publication_unique_passed": 400, + "focused_passed": 11, + "binary_passed": 2, + "directory_passed": 12, + "directory_failed": 1, + "python_passed": 96, + "counts_exclude": "focused reruns and nested subprocess summaries; preceding-source isolated reruns are historical evidence only", + "release_qualified": false, + "remote_preceding_head_ci": [ + { + "head": "6101d5869f71dd836876266ba8d5aec0ac7965b0", + "run": 37241921975, + "rust_job": 111552187935, + "conclusion": "failure", + "server_library_passed": 693, + "binary_passed": 2, + "directory_passed": 12, + "directory_failed": 1, + "failure": "directory_reservations_recover_two_distinct_repository_cells: retired ingestion command descriptor is unavailable", + "log": { + "path": "/tmp/canopy-pr34-6101d58-failed.log", + "sha256": "a89ae34ea371c2a33e537ccfa7d32febce559acb2fbabea896bc639d10ea5900" + } + }, + { + "head": "6101d5869f71dd836876266ba8d5aec0ac7965b0", + "run": 37241919745, + "rust_job": 111552180632, + "conclusion": "failure", + "server_library_passed": 693, + "binary_passed": 2, + "directory_passed": 12, + "directory_failed": 1, + "failure": "directory_reservations_recover_two_distinct_repository_cells: retired ingestion command descriptor is unavailable", + "log": { + "path": "/tmp/canopy-pr34-6101d58-secondary-failed.log", + "sha256": "52cfc5d915246bdee2a28a824278713a3f8377241204a91018284e69b15fc7ed" + } + } + ], + "failed_drafts": [ + { + "reason": "new fixture observed Bound as a terminal phase instead of waiting for the actual expiry fence", + "validation": { + "source_files": 488, + "rust_files": 474, + "source_hash_digest": "39b90a9b5aafabc8d5273d314105690a9ee943f883e350d797bba348f10270b2", + "release_qualified": false, + "execution_complete": false, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 0.963, + "log": "/tmp/canopy-stage-physical-clippy-final.log", + "log_sha256": "a7642a0867425c6137f6fda230ee0c5445d5148745ccd7b4f5b0b4f2b4540d91", + "summaries": [], + "failed_cases": [] + }, + { + "label": "focused", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "physical_creating_worker", + "physical_bound_worker", + "git_http::capture::tests::", + "native_receive_stages_verifies_and_publishes_then_clones_after_cache_loss", + "native_receive_prepares_bounded_immutable_root_completion_from_registered_custody", + "--test-threads=4" + ], + "exit_code": 101, + "seconds": 108.937, + "log": "/tmp/canopy-stage-physical-focused-final.log", + "log_sha256": "eabbdba76d268803f8da2bc5e1dc7cafc4274155d67a4a693c3284e7369876bd", + "summaries": [ + [ + 5, + 1, + 0, + 0, + 689 + ] + ], + "failed_cases": [ + "packs::publication::tests::staging_service::physical::physical_bound_worker_outlives_custody_fence_async_abort_and_observer_drop" + ] + } + ] + }, + "validation_file": { + "path": "/tmp/canopy-stage-physical-before-fence-wait-validation.json", + "sha256": "067facbdc2f16ec424fbb918608d7fdf88e704b59ba1527de8ec4e12062061df" + } + }, + { + "reason": "the old phase fixture retained its newly owned context across Bind; an unrelated serving timeout passed in exact isolation", + "validation": { + "source_files": 488, + "rust_files": 474, + "source_hash_digest": "d7b3210e3a0011a39eb1aad7c257ceb6fc7490752e52c59edf21bb8df08a1e9b", + "release_qualified": false, + "execution_complete": false, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 51.379, + "log": "/tmp/canopy-stage-physical-clippy-final.log", + "log_sha256": "376d4d7ca543747510c2717416f9c17db2c89cfd726d864a4bc5270edba5175a", + "summaries": [], + "failed_cases": [] + }, + { + "label": "focused", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "physical_creating_worker", + "physical_bound_worker", + "git_http::capture::tests::", + "native_receive_stages_verifies_and_publishes_then_clones_after_cache_loss", + "native_receive_prepares_bounded_immutable_root_completion_from_registered_custody", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 118.525, + "log": "/tmp/canopy-stage-physical-focused-final.log", + "log_sha256": "8b2d825152bcc50f2364b87e033da72fd0b1e937c3ec5c33cbc9e0bf31d1c7d6", + "summaries": [ + [ + 6, + 0, + 0, + 0, + 689 + ] + ], + "failed_cases": [] + }, + { + "label": "libraries", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked" + ], + "exit_code": 101, + "seconds": 261.895, + "log": "/tmp/canopy-stage-physical-libraries-final.log", + "log_sha256": "24a23f9ccccd15fdc51cff98a3a8327b4cf6ce14172c6fb33c1e5fffb6a55c44", + "summaries": [ + [ + 6, + 0, + 0, + 0, + 0 + ], + [ + 14, + 0, + 0, + 0, + 0 + ], + [ + 1, + 0, + 0, + 0, + 694 + ], + [ + 1, + 0, + 0, + 0, + 694 + ], + [ + 693, + 2, + 0, + 0, + 0 + ] + ], + "failed_cases": [ + "packs::publication::tests::serving::lifecycle::closed_owner_keeps_borrowed_generation_renewing_until_last_snapshot_clone_drops", + "packs::publication::tests::staging_service::bound::bound_service_phase_handoff_and_residence_cap_fence_existing_bases_and_inflight_workers" + ] + } + ] + }, + "validation_file": { + "path": "/tmp/canopy-stage-physical-before-context-release-validation.json", + "sha256": "6b4d88f8b900addaf95f5163c0115b25d70027c3c5a715984f9d86e8936f3b36" + } + }, + { + "reason": "the one-second test ceiling sometimes expired during Bind setup; same limit is now applied after Bind before the publication scenario", + "validation": { + "source_files": 488, + "rust_files": 474, + "source_hash_digest": "56c289487ff8c16d5b1640e6ede2ce66a5a1b56232eeda0ea10daf4bebf1072c", + "release_qualified": false, + "execution_complete": false, + "phases": [ + { + "label": "clippy", + "command": [ + "cargo", + "+1.98.0", + "clippy", + "--workspace", + "--all-targets", + "--locked", + "--", + "-D", + "warnings" + ], + "exit_code": 0, + "seconds": 43.779, + "log": "/tmp/canopy-stage-physical-clippy-final.log", + "log_sha256": "8157545edd7feebc907ecf0ac53538f3288bb13543ef8e50ef3ad9cbb1d4aa52", + "summaries": [], + "failed_cases": [] + }, + { + "label": "focused", + "command": [ + "cargo", + "+1.98.0", + "test", + "-p", + "canopy-server", + "--lib", + "--locked", + "--", + "physical_creating_worker", + "physical_bound_worker", + "git_http::capture::tests::", + "native_receive_stages_verifies_and_publishes_then_clones_after_cache_loss", + "native_receive_prepares_bounded_immutable_root_completion_from_registered_custody", + "bound_service_phase_handoff_and_residence_cap_fence_existing_bases_and_inflight_workers", + "closed_owner_keeps_borrowed_generation_renewing_until_last_snapshot_clone_drops", + "--test-threads=4" + ], + "exit_code": 0, + "seconds": 86.05, + "log": "/tmp/canopy-stage-physical-focused-final.log", + "log_sha256": "a3e58e61ad3fa19d2d97b74b3403c93c20fc3a8e294436ac18dc5eaf8b349c52", + "summaries": [ + [ + 8, + 0, + 0, + 0, + 687 + ] + ], + "failed_cases": [] + }, + { + "label": "libraries", + "command": [ + "cargo", + "+1.98.0", + "test", + "--workspace", + "--lib", + "--locked" + ], + "exit_code": 101, + "seconds": 218.733, + "log": "/tmp/canopy-stage-physical-libraries-final.log", + "log_sha256": "97848e431c678c2655439767556515e9d41c7da966f738c343a10ebe861e6918", + "summaries": [ + [ + 6, + 0, + 0, + 0, + 0 + ], + [ + 14, + 0, + 0, + 0, + 0 + ], + [ + 1, + 0, + 0, + 0, + 694 + ], + [ + 1, + 0, + 0, + 0, + 694 + ], + [ + 694, + 1, + 0, + 0, + 0 + ] + ], + "failed_cases": [ + "packs::publication::tests::staging_service::publication::bound_final_ceiling_discards_held_proof_and_drops_result_before_worker_credit" + ] + } + ] + }, + "validation_file": { + "path": "/tmp/canopy-stage-physical-before-ceiling-phase-validation.json", + "sha256": "8311028376a090240019592b2cb3e29986b4cf289e849185dcd54ce055242a5e" + } + } + ], + "isolated_draft_runs": [ + { + "path": "/tmp/canopy-stage-physical-isolated-failures.log", + "sha256": "124d50cd9af13b02685fef4453139f8bdef7f25d52a5e1c9573e5f6c11f33160" + }, + { + "path": "/tmp/canopy-stage-physical-ceiling-isolated.log", + "sha256": "1d6b30e98ad6bc6efd1acfaf57cb0621dafd22493eb579e892bcdac79c347936" + } + ], + "initial_compile_diagnostic": { + "path": "/tmp/canopy-stage-physical-clippy-draft.log", + "sha256": "14ac3609200cff14279207a03ac327991f4dc4583cefbd660f80bda05a1b81bf" + }, + "initial_compile_diagnostic_scope": "pre-qualification moved-value and result-alias mistakes; no passing qualification is attributed to this draft", + "remaining_ci_failure": "The real write path and directory fixture still use retired loose-object ingestion. Do not restore that command/schema or skip the case; convert the actual native producers before their fixtures.", + "unrun": [ + "complete latest-source workspace command beyond the directory failure", + "remaining integration binaries and doctests", + "full Linux/RustFS provider workflow", + "full histories, OS containment and 10000-engineer mixed-load/recovery capacity", + "adopted old-owner inputs through authenticated retained selection" + ], + "next_priorities": [ + "resident StagingCoordinator ownership and actual HTTP/SSH/generated producers", + "request/metadata/policy physical-owner propagation and bounded persisted metadata replay using existing structures", + "mandatory joint-root policy/outcome completion and real consumer fixture conversion", + "full Linux/provider CI, owner-aware authoritative consumers and physically fenced adoption", + "full remaining hard-cutover, custody history, GC/backup/restore, resource containment, native maintenance/hot-cache, capacity and attribution scope" + ], + "protected_index_sha256": "bef77b0a83f80518f232060828e83797174b1863b8ed9147bffa65850af59798", + "protected_archive_sha256": "c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e" +} diff --git a/docs/implementation.md b/docs/implementation.md index 20edc7c9..c84097c2 100644 --- a/docs/implementation.md +++ b/docs/implementation.md @@ -89,7 +89,7 @@ See [native pack policy](contracts.md#native-pack-resource-policy) and ### Local workspace and shutdown -The node locks its `data_dir` and owns `runtime-v1/` beneath it. On Unix, +The node locks its `data_dir` and owns `canopy-pack-v1/` beneath it. On Unix, restart removes abandoned local state before restoring Cells from object storage; live Git descendants prevent cleanup. Unknown runtime markers and cleanup errors stop startup. Keep the lock files in place; files outside the managed runtime diff --git a/docs/large-repository-implementation-plan.md b/docs/large-repository-implementation-plan.md index 0608711f..17a52499 100644 --- a/docs/large-repository-implementation-plan.md +++ b/docs/large-repository-implementation-plan.md @@ -91,7 +91,7 @@ Populate it from the admitted actor authority. Audit replay and retry execution **Files:** `crates/canopy-server/src/schema.sql`, `crates/canopy-server/src/lib.rs`, `crates/canopy-server/src/packs/{metadata,directory,sources,catalog,verification,closure}/`, new operation/publication commands, existing ref/push/product commands and repository Cell tests. **Dependency:** B's actual admitted owner fence; C's authenticated artifacts; E's complete isolated physical verification. -The earlier [SQL fixture](design/packed-repository-schema.sql) is not the release definition. Deliver fresh release DDL together with the new deployment marker. Preserve product tables unless a demonstrated requirement changes them; keep push response/certificate chunks. Remove historical Git bodies, mutable per-object placement and graph rows from the Cell. Store current catalog descriptor/generation, root certification, fenced operation attempts/outcomes and retention/reader-pin facts. Immutable metadata segments and directory/source indexes hold canonical object/edge inventories. Do not implement the superseded `PutObjects`, `CertifyObjects` or `SwitchPackLocations` Cell commands as a compatibility stage. +The earlier [SQL fixture](design/packed-repository-schema.sql) is not the release definition. Deliver fresh release DDL together with the new deployment marker. Preserve product tables unless a demonstrated requirement changes them; reuse the existing immutable response, signed-certificate and native-result roots instead of inline SQL body/chunk adapters. Remove historical Git bodies, mutable per-object placement and graph rows from the Cell. Store current catalog descriptor/generation, root certification, fenced operation attempts/outcomes and retention/reader-pin facts. Immutable metadata segments and directory/source indexes hold canonical object/edge inventories. Do not implement the superseded `PutObjects`, `CertifyObjects` or `SwitchPackLocations` Cell commands as a compatibility stage. 1. Reuse the implemented immutable artifacts, canonical `ObjectHeader`/typed edges, native index ordinal partitions and bounded catalog codecs. Operation records bind repository, format, identity, admitted owner fence/attempt, input catalog/generation, artifact descriptors and exact durable response. Register large input/output sets through a bounded immutable root; never serialize every historical descriptor into Begin or Complete. 2. Implement a trusted base resolver from the retained, certified input catalog and authoritative root facts. Keep a valid generation lease throughout preparation. Reuse `CatalogFiles` for admitted, authenticated SQLite files and `CatalogReader::headers` for grouped file reads. Resolve only requested OIDs in ordered batches of at most 512; presence in a native index or raw catalog is insufficient for closure certification. Compare the complete incoming canonical header against every matching base header, including body and graph digests. @@ -156,6 +156,7 @@ The earlier [SQL fixture](design/packed-repository-schema.sql) is not the releas 2. For generated candidate validation, compare expected canonical OID/body digest/size and certified metadata. Preserve ordered commit parents and exact policy inputs; unordered `commit_parents` is not an order proof. 3. Replace ancestry's in-memory discovered-commit limit and permanent mutable SQL ancestry projection with certified immutable commit metadata/native commit graphs plus admitted disk-backed traversal scratch when needed. Reuse existing OID/typed parent meanings and verify every selected path against the pinned certified catalog. Bind a bounded ancestry certificate to the exact old/new OIDs, canonical inventories and publication context; authenticate it in the final policy transaction. Do not submit the legacy parent-proof command to tables removed by the fresh schema. Keep cancellation and admission; incomplete traversal is an error, not a negative result. The fallback now reuses admitted SQLite growth, exclusive per-walker traversal, exact StoredCatalog memo binding and permanent failure/cancellation fencing. Queue resets page at most 512 keys and preserve bounded exact-catalog answers. Native commit-graph acceleration, serving/candidate integration and native histories exceeding 100k commits still require implementation/qualification. 4. Audit minimum receipts and ref snapshots across product reads after owner movement. Do not read a locally cached newer/older branch in place of the selected authoritative snapshot. +5. Implement asynchronous, commit-pinned directory entry attribution through the [file attribution design](design/file-attribution.md). Qualify exact per-path Git merge semantics, bounded cold history jobs and cache/generation ownership before serving results. Reuse the immutable index infrastructure if measured workloads require persistent attribution; do not add per-file/per-commit SQL storage or put historical backfill on the push critical path. **Acceptance:** a synthetic history exceeding 100k commits can check positive/negative ancestry, prepare merge/rebase candidates and exercise branch protection within configured budgets; wrong parent order/body digest is rejected; wide tree pagination and binary paths remain correct. Signed commits/tags and SHA-256 candidates pass existing semantics. @@ -171,7 +172,7 @@ The earlier [SQL fixture](design/packed-repository-schema.sql) is not the releas **Shared publication dispatch implemented:** `ready_compaction` issues the existing maintenance certificate and retains the exact SDK command. `PublicationCoordinator` admits push/compaction variants with typed outcomes, reserved class counts/encoded bytes, per-class actor counts, bounded foreground bursts and maintenance concurrency. Both classes share cancellation-safe retention, supervision, pending lookup, original-receipt resolution and close/drain. Defaults reserve four maintenance operations and two of eight durability waits; the geometric native fixture now uses this dispatcher for every publication. Continuous preparation, whole-process CPU/I/O shares, renewal/reaping, durable reconstruction and production invocation remain required. See the [shared dispatch contract](design/shared-publication-dispatch.md). -**Files:** new `crates/canopy-server/src/packs/maintenance.rs`, D's commands, existing `crates/canopy-server/src/git_gateway/maintenance.rs`, node maintenance/admission scheduling. +**Files:** new `crates/canopy-server/src/packs/maintenance.rs`, D's commands, owned native workspaces in `crates/canopy-server/src/packs/publication/serving/session/`, and node maintenance/admission scheduling. The legacy shared-loose-cache repack loop is retired; its replacement must schedule native generations under the existing fair admission and physical pins. 1. Begin a durable compaction operation and pin its selected catalog generation and exact input descriptors. Initial selection: at most 32 packs or 8 GiB compressed; prioritize small/duplicate packs, leave large stable history alone. A single over-budget pack requires a separately admitted job, not an unbounded default repack. Account for directory-run compaction separately from physical pack rewriting. 2. Stream selected preferred objects into structural/blob OID spools. Pack each with native settings from the design, verify and upload outputs. Preserve all canonical objects, including unreachable ones. @@ -192,7 +193,7 @@ The earlier [SQL fixture](design/packed-repository-schema.sql) is not the releas 1. Add a read-only retained-root enumeration API to Cellule. Reuse existing root/pin types. Return paginated roots with retention reason, control/pin revision bindings and a completion indicator. A caller may treat it as exhaustive only while its existing maintenance barrier prevents root/pin changes; detect revisions changing during enumeration and fail/restart. 2. Audit every recovery selector, pin and unfinished backup/restore path. Test that each possible selected root appears in the enumeration. If any cannot be enumerated, return an explicit incomplete result; the collector must refuse deletion. 3. Canopy walks each retained SQL snapshot's current catalog, immutable generation facts, independent preparation pins/checkpoint certificates, unfinished/uncertain operation outcomes and LFS facts. Resolve each retained catalog's complete authenticated directory/source dependencies, including immutable index nodes and all required metadata/pack/index artifacts. Preserve private attempt namespaces and unregistered output ownership until their writers are proven drained; never infer their deletion eligibility solely from catalog membership. The fresh schema has no `objects.pack_id` inventory. Retired catalog tombstones and completed operations alone do not retain bytes. Include all certified canonical objects, not only currently referenced tips. Restore preserves logical IDs, descriptors and the artifact allocation watermark; an isolated rollback restore uses a new provider namespace so delayed old deletes cannot target its bytes. -The [terminal recovery protocol](design/terminal-publication-retention.md) now transfers the selected completed attempt's original certificate/journal and release receipt into the existing immutable push row and frees its independent pin atomically. Its typed closed audit verifier retains header/history, selected response and native plan/signed audit edges. This supplies selected-attempt retirement, not complete root enumeration or artifact deletion. Include these archived rows and exact historical receipts in retained-root traversal and isolated backup/restore. Older unknown attempts keep their own pins until the required recovery/retention protocol proves their disposition. +The [terminal recovery protocol](design/terminal-publication-retention.md) now transfers the selected completed attempt's original certificate/journal and release receipt into the shared immutable receipt archive keyed by original incarnation/admission sequence and frees its independent pin atomically. Its typed closed audit verifier retains header/history, selected response and native plan/signed audit edges. This supplies selected-attempt retirement, not complete root enumeration or artifact deletion. Include these archived rows and exact historical receipts in retained-root traversal and isolated backup/restore. Older unknown attempts keep their own pins until the required recovery/retention protocol proves their disposition. 4. Copy and verify all required artifacts before setting backup complete. Preserve/restore product tables and push replay state as in existing backup behavior; no old-format migration is added. @@ -292,8 +293,8 @@ Each cell below is a test family, with SHA-1/SHA-256 coverage on a representativ The [immutable ref state](design/immutable-ref-state.md) supplies conditional versioned roots and streaming initial construction. Complete the final publication change in this order: 1. The shared streaming rewrite now coalesces existing-base batches by affected subtree and preserves untouched roots. Qualify sustained ordinary and bulk preparation separately, including long-name byte splits, retained tombstones, provider budgets and hot-root fairness. -2. The fresh immutable catalog generation now carries the ref snapshot through the same query-derived base and retention floor. The private preparation factory loads and rewrites that exact root; compaction carries it forward and the inline publisher refuses selected roots. Fresh empty initialization now authenticates a private empty preparation and atomically installs joint roots with one durable outcome; wire it into repository creation at cutover. Bind membership/ancestry and exact policy/check facts to the privately issued transition certificate, and qualify their current-state CAS/fairness semantics. -3. Direct-push [paged policy guards](design/paged-ref-policy-guards.md) now bind rare configuration epochs and indexed exact check dependencies, with bounded transactional registration/cleanup and private conditional root signing. The private [immutable completion factory](design/immutable-push-outcomes.md) binds registered native custody and freezes success/refusal descriptors in an 8 KiB input. Command 36 atomically publishes those exact catalog/ref/response descriptors with live guard/epoch, current authorization, owner/lease/pin and generation CAS checks; it transports no plan and writes no per-ref rows. Query 37 and the streaming replay adapter derive the selected result from current authorized durable identity. The service-owned exact command factory, foreground dispatch and bound lifecycle now retain/recover this command with a 16 KiB wire reservation and current-authorized streaming ticket responses. Immutable outcome-only command 38 reuses the session certificate, native checkpoint, same RootPush dispatch and current-authorized streaming reader without catalog/refs writes. Paged-policy commands now retain their original intent/evidence and exact SDK identity in the same foreground dispatcher; the bound lifecycle resumes after a known page and blocks final handoff during uncertainty. Armed pages now retain a pre-frozen refusal-only command under one 528 KiB admission and recover the original SDK evidence for the active local phase. Successful pages share one refusal Arc; that exact command can also complete a negative outcome after later policy/write changes. Its final transaction cannot select native success or expose roots. Complete durable process-loss reconstruction, production HTTP/SSH refusal-pipeline conversion and reviewed-merge bindings, and include every page/command cost in hot-repository capacity qualification. +2. The fresh immutable catalog generation now carries the ref snapshot through the same query-derived base and retention floor. The private preparation factory loads and rewrites that exact root; compaction carries it forward and the inline publisher refuses selected roots. Fresh empty initialization authenticates a private empty preparation and atomically installs joint roots with one durable outcome. The local production cutover now invokes that path before repository Ready and verifies the retained immutable initialization during fresh-disk restore. Final initialization now uses the existing exact registered snapshot/body/journal protocol, including cold original denial recovery. Typed initialization retirement now reuses that shared archive and releases the selected initial pin during actual startup, while preserving original positive/negative receipts. Complete initial Begin/Claim/Renew, pre-registration process-loss recovery and background recovery of older orphan attempts before release. See the [current qualification and unreleasable boundaries](large-repository-implementation-status.md#production-hard-cutover-started-locally). Bind membership/ancestry and exact policy/check facts to the privately issued transition certificate, and qualify their current-state CAS/fairness semantics. +3. Direct-push [paged policy guards](design/paged-ref-policy-guards.md) now bind rare configuration epochs and indexed exact check dependencies, with bounded transactional registration/cleanup and private conditional root signing. The private [immutable completion factory](design/immutable-push-outcomes.md) binds registered native custody and freezes success/refusal descriptors in an 8 KiB input. Command 36 atomically publishes those exact catalog/ref/response descriptors with live guard/epoch, current authorization, owner/lease/pin and generation CAS checks; it transports no plan and writes no per-ref rows. Query 37 and the streaming replay adapter derive the selected result from current authorized durable identity. The service-owned exact command factory, foreground dispatch and bound lifecycle now retain/recover this command with a 32 KiB registered wire reservation and current-authorized streaming ticket responses. Immutable outcome-only command 38 reuses the session certificate, native checkpoint, same RootPush dispatch and current-authorized streaming reader without catalog/refs writes. Paged-policy commands now retain their original intent/evidence and exact SDK identity in the same foreground dispatcher; the bound lifecycle resumes after a known page and blocks final handoff during uncertainty. Armed pages now retain a pre-frozen refusal-only command under one 544 KiB registered admission and recover the original SDK evidence for the active local phase. Successful pages share one refusal Arc; that exact command can also complete a negative outcome after later policy/write changes. Its final transaction cannot select native success or expose roots. Complete durable process-loss reconstruction, production HTTP/SSH refusal-pipeline conversion and reviewed-merge bindings, and include every page/command cost in hot-repository capacity qualification. 4. Convert every producer, reader, default-branch, policy/check, review and recovery path together. Delete the old ref/body schema and adapters for the fresh-data cutover. 5. Include snapshots and their transitive immutable nodes in complete retention, collection and isolated restore, including the immutable initialization outcome's retained empty catalog/ref roots. Qualify hot-root fairness and full-history mixed workloads against the mandatory large-team gates. diff --git a/docs/large-repository-implementation-status.md b/docs/large-repository-implementation-status.md index 7bc536be..7a5d14b2 100644 --- a/docs/large-repository-implementation-status.md +++ b/docs/large-repository-implementation-status.md @@ -1,10 +1,843 @@ # Large-repository implementation status -Updated during implementation on 2026-10-03. **The full implementation and capacity goal remains open.** The [large-team amendment](large-team-scalability.md) is mandatory scope alongside the original storage design. Passing primitive tests is not completion of the hard cutover or proof of capacity. +Updated during implementation on 2026-10-04. **The full implementation and capacity goal remains open.** The [large-team amendment](large-team-scalability.md) is mandatory scope alongside the original storage design. Passing primitive tests is not completion of the hard cutover or proof of capacity. + +Current cutover review: [PR #34](https://github.com/crabbuild/canopy/pull/34), +directly against `main`. The original SQL hydration failures have been resolved by +converting their real read/cache callers. Full CI remains open: the directory +integration fixture still invokes the retired loose-object ingestion command. +The physical worker ownership change below supplies a prerequisite for the native +write replacement; it does not complete that replacement. The PR remains for +review and is not ready to merge or deploy. Older checkpoint notes describe +historical states. + +Implementation is isolated in the PR worktree. The original checkout contains an unrelated, extensive staged workspace merge; its workspace, benchmark and runtime work has been preserved. Canopy is split into Git-format, object-storage and server crates. Main now contains all completed PR #20–#30 changes through [PR #31](https://github.com/crabbuild/canopy/pull/31), merged at `db80fd836db94fff894030f02d736fe92840748c`. The PR #31 checkpoint audit verifies each directly merged PR's exact merge tree and main ancestry; that checkpoint's entire tree is identical to completed PR #30 (`5bf48677857e3d1dd769aa7f1d73eb5db00db30f`). PRs #28–#30 originally merged into stack branches and reached main through #31. Both #31 Verify runs, [37132349361](https://github.com/crabbuild/canopy/actions/runs/37132349361) and [37132329706](https://github.com/crabbuild/canopy/actions/runs/37132329706), pass harness and Rust. The merged main revision also passes [Verify 37132672371](https://github.com/crabbuild/canopy/actions/runs/37132672371). + +All five Cellule dependency declarations and six lockfile entries pin `161067f5a21703b3e257024bcb64e565fd9657b4` from [Cellule PR #50](https://github.com/crabbuild/cellule/pull/50), including the admitted owner fence, exact-command snapshot and admitted-mutation APIs. Historical validation below remains attributed to its original source revisions. Trusted ref-plan certification, typed catalog/ref publication, immutable exact-response completion and the class/account-fair dispatcher exist. The local production cutover now selects their registry/schema and initializes new repositories through certified empty roots. Production HTTP/SSH/generated producers and authoritative readers, complete startup recovery and the final schema hard cutover remain open. + +## Staging physical worker ownership + +Staging contexts and result slots now share the original worker admission. +Result transfer, async cancellation and custody expiry cannot release worker or +actor capacity while detached physical work retains it. Bound callbacks receive +the live preparation session and its staging context. The existing `ReadOwner`, +cache ownership and admitted spool structures carry the pin; there is no new +queue, durable schema or independent physical-worker inventory. + +Native receive requires the checked context and carries its owner in the Git +process group and response reconciliation. Capture enumeration and authenticated +pack/index upload carry the same owner in blocking jobs and pinned files. +`PhysicalVerifier::download_staged` validates live repository, current creating +namespace and format; cache construction/download, native index validation, +canonical decoding, edge-spool writes and deferred cleanup retain the worker. +Inspection and witness completion check live custody again. The unowned download +is confined to qualification builds. The native-receive qualification now runs +physical verification as an admitted producer rather than outside the lifecycle. A phase-handoff +fixture now releases its transferred worker context before Bind and checks that +the retained staging token cannot reopen staging afterward. Holding that context +across Bind would correctly keep physical admission charged. + +This is a lifecycle/API prerequisite. Live HTTP/SSH/generated producer wiring, +resident staging service ownership, old-owner input adoption, bounded metadata +replay and final joint-root completion still need conversion. In particular, +current-namespace verification does not authorize adopted inputs from an older +attempt; those need the authenticated retained-input selection path. Request, +metadata and policy workers outside these native stages still need the same +physical-owner integration. Full CI and capacity qualification remain open. + +Final frozen-source qualification passes all 715 unique workspace library cases +(6 Git-format, 14 object-storage, 695 server), including all 400 publication +cases, and all 11 focused ownership/native/expiry cases. Two binary cases, the +server build, all-target workspace Clippy with warnings denied, formatting/diff +checks and all 96 Python harness cases pass. Directory integration remains 12 +passes and one failure in retired ingestion. Thus 730 unique Rust cases executed, +729 pass and one fails; focused reruns and two nested subprocess summaries are +excluded. Later integration binaries/doctests and Linux/provider/capacity gates +remain unrun for this source. Both actual preceding `6101d58` CI runs pass all +693 server library cases and fail at the same directory case. Failed draft and +isolated diagnostics, exact source/commands/log hashes and limits are preserved +in the [physical-worker evidence](evidence/staging-physical-workers-20261004.json). +The transferred-context fixture releases admission before Bind and checks the +old token's staging denial afterward. One-second publication ceilings are now +applied after actual Bind before private ready-proof construction, exercising +the same ceiling guard without making startup speed a test prerequisite. The +ceiling duration, virtual-time advance and publication/result/credit assertions +remain unchanged. One preceding-source serving timeout passes in exact isolation +and the final full rerun; no claim of eliminating all timing failures is made. + +## Immutable source cursor and native write-base conversion + +The private cache constructor used by HTTP push and generated candidate +preparation now takes one retained joint snapshot. It reads only requested ref +expectations in batches of at most 128 names and 256 KiB, including tombstone +versions; the native baseline streams all live refs directly into packed-refs. +Post-receive comparison resolves exact requested names with owned native cat-file +batches (32 names / 64 KiB input), avoiding prefix enumeration and changes to +unrequested refs. Candidate preparation requests no repository-wide ref map. + +Catalog inputs now stream from the immutable source index in count/descriptor- +byte-bounded pages. The shared range-tree difference cursor compares full records +at matching first keys, emits additions/replacements, omits deletions and skips +identical authenticated subtrees. Fixed old/new roots exclude later publications; +new lower keys are found by starting a new root-pair difference, rather than +continuing an OID or retired SQL sequence. A byte-excluded descriptor is retained +for the next page. Cancellation/error poisons the cursor; callers restart with +the same root pair after their last returned key. Its two frontier stacks retain +only bounded-height paths and siblings. Source shard descriptors reuse the existing +range-tree structure and physical pack bindings. + +Native write-base construction currently enumerates `None -> selected sources`: +it is a cold per-request baseline, not a coalesced incremental cache. Admitted +disk stores physical pack deduplication. Authenticated pack/index inputs remain +native, and a separate writable sibling cache uses that baseline as an alternate. +This prevents accepted catalog packs from being mistaken for new receive outputs. +Both cache owners retain the snapshot, physical generation and cleanup admission; +construction refreshes current read authority and lease observations and rechecks +before returning. This internal base grants neither fetch reachability nor mutation +publication authority. Fetch continues to require its distinct completed closure. + +The retired sequence/high-water hydration caller, shared loose-cache reuse and +its obsolete periodic repack loop are removed. Unused loose-object/repack helpers +are confined to unit-test fixtures to preserve native cache mechanism coverage. +Legacy object ingestion, graph preparation, candidate reservation/completion and +HTTP/SSH write publication still require their owned staged-producer conversion. +Some legacy object APIs remain genuine callers and are not relabeled as converted. +Native generation maintenance, workspace sharing and cold-history performance +qualification remain open. + +Nine focused replacement cases pass: lower-key/future-root isolation, byte-prefix +restart, duplicate/failed/deleted/replaced records, subtree-read bounds after +10,000 sources, independent differential results across tree shapes, native +baseline/incoming-pack isolation, literal ref lookup, and real provider +cancellation/revocation through drain. The focused command takes 114.256 seconds +(including compilation); test runtime is 2.70 seconds. A failed draft cancellation +test suspended at ref-root loading before file admission; it is preserved in the +evidence. Warming that immutable metadata moves the gate to its intended native +pack-transfer phase without relaxing the physical retention assertion. + +The exact full workspace command runs all 713 library cases successfully, +including 693 server cases and all 398 publication cases, plus two binary cases. +It then passes 12 directory integration cases and fails +`directory_reservations_recover_two_distinct_repository_cells` with +`Registry("operation descriptor is unavailable")`: its fixture still invokes the +retired loose-object ingestion command. Later integration binaries and provider +qualification have not run. Nine selected portable workspace/startup/drain cases +pass independently, including production packed-repository initialization. +Total executed Rust coverage is 737 unique cases: 736 pass, one fails; focused +reruns and two nested subprocess summaries are excluded. Linux-only fork cases +were not run locally. This is progress on CI, not a green full workflow. + +Workspace/all-target Clippy with warnings denied (36.465 seconds), server build +(45.072 seconds), formatting, diff checks and all 96 Python harness cases pass. +The protected original index/archive, clean read-only SDK and exact dependency +pins are unchanged. Frozen source includes 487 files / 473 Rust files; its +fingerprint, commands, log digests and preceding CI diagnostics are in +[evidence](evidence/serving-native-write-base-20261004.json). + +Highest next priorities are the real owned HTTP/SSH/generated write producers, +then integration-fixture conversion against those producers and complete +workflow/provider qualification. Authority, recovery, custody history/rollover, +final DDL, GC/backup/restore, resource containment, fair native maintenance, +workspace sharing, full-history/team capacity and file attribution remain +mandatory scope. + +## Certified native transport workspace checkpoint + +Local HTTP and SSH fetch now use complete native forward workspaces from an +accepted joint snapshot. Live refs stream from the immutable ref index in +32-name pages, including the same default branch, without a history-sized ref +map or the explicit-object-root cap. An admitted SQLite spool tracks typed +frontier, completed membership and distinct physical pack inputs. Each selected +pack/index pair is authenticated, installed once and physically verified; +blobs remain packed. Frontier batches have at most 128 objects and edge/ +membership pages have at most 512. Physical pack presence cannot authorize an +unreferenced want. HTTP and SSH validate every wanted OID against the retained +refs' completed closure before forwarding it to Git. + +Construction retains a producer borrow and physical generation guard, refreshes +exact lease/owner/access observations during bounded steps, and rechecks before +returning. Renewal can carry construction past its initial read deadline; an +expired pin cannot be resurrected. Cancellation detaches observers while actual +provider/blocking/native work and descendants remain owned. A completed workspace +releases construction read credit while retaining file/disk admission and its +snapshot until physical cleanup. The two-credit regression covers that lifetime +separation. + +All 17 focused cases pass (107.869 seconds command, +13.14 seconds test runtime). They cover both object formats, actual HTTP clones, +complete 532-parent native history, unrelated physical objects, 10,000 streamed +ref names, admission/size/type refusals, cancellation, revocation, expired pins, +and real drain. Fixtures physically verify native packs then install joint facts +through trusted SQL; they qualify consumers, not live write publication. Existing +lease tests now use service-owned renewal during expensive fixture construction, +observed renewal milestones, and short real-clock expiry injected at the intended +recovery phase. No production lease duration is increased. + +The frozen workspace library runs 708 unique cases: +704 pass and four legacy `object_reads` cursor/rollback/byte-bound cases +fail on `no such table: objects`. All 396 publication cases +pass. Nine selected portable workspace/lifecycle cases pass, giving +717 unique Rust cases, 713 +passes and four failures; focused reruns and nested subprocess summaries are not +counted twice. Workspace/all-target Clippy with warnings denied passes +(37.712 seconds), as do server build, formatting, diff checks and all +96 Python harness tests. The driver preserves the actual library exit 101. +Linux-only fork cases and the full workspace integration/provider campaign were +not run locally. Source fingerprint, command results, log digests and earlier +failed diagnostic runs are in +[workspace evidence](evidence/serving-certified-workspaces-20261004.json). +Protected original index/archive and exact SDK pins remain unchanged. + +Highest next priority is replacing legacy write/cache object-page readers and +their real callers with immutable catalog/native APIs, preserving paging, +rollback and bounded incremental-work coverage. Then finish owned HTTP/SSH/ +generated write producers, live pull/editorial ref authority and owner-aware +remote routes. Current transfers build their own workspace and download inputs; +coalescing authorized generation workspaces, bulk graph scheduling and cold I/O +remain required for large-repository latency. Physical owner adoption, custody +history/rollover, final DDL, GC/backup/restore, OS containment, native maintenance, +signed completion, attribution and full-history/10,000-SDE capacity qualification +remain mandatory. This checkpoint does not establish capacity or latency targets. + +## Certified browser objects and ancestry checkpoint + +Local directory, file, tag and first-parent history views now borrow one accepted +joint catalog generation for the whole request. Object presence, kind and size +come from certified headers; bounded canonical bodies come from the shared +verified native pack service. The reader no longer queries legacy `objects`, +`object_closure`, `object_edges` or `commit_parents` tables. The existing raw path, +mode, pagination and preview behavior remains. Zero/absent revision IDs return +missing; wrong-format IDs reject as invalid. + +Comparison files, patches and previews also use this snapshot for their objects +and merge-base graph. The new typed edge page accepts at most 128 sorted parent +IDs and returns at most 512 edges with a tuple continuation. Preferred metadata +queries run on admitted blocking work retaining a child physical guard. The +shared private read contract rechecks current access/owner/lease and retains +workers after observer cancellation. Ancestry filters commit edges, consumes all +pages and keeps the existing bounded graph budget without a native process per +commit. Equal tips still require certified commit membership. Live pull/review/ +thread editorial metadata and ref authority remain unconverted. + +Six new families exercise real HTTP reads in both object formats, 32-entry tree +and history continuation, annotated tags, raw non-UTF-8 and literal paths, +executable files, symlinks, gitlinks, bounded/binary previews, patches, merge bases, +cached generation isolation, access revocation, and canceled suspended-provider +work. A native 532-parent merge verifies a relevant parent beyond the first +512-edge page. Fixtures physically verify actual native packs and metadata; +their joint catalog facts and editorial records are installed by trusted SQL to +isolate consumers. They do not qualify live producer/publication or capacity. + +All 112 focused cases pass (35.01 seconds test runtime; +120.624 seconds command including compilation). Warnings-denied +workspace/all-target Clippy passes in 31.581 seconds. +The workspace library runs 695 unique cases: +690 pass and the same five legacy object readers fail. All +387 publication cases pass. Nine selected portable +workspace/lifecycle cases pass, giving 704 unique Rust +cases, 699 passes and five failures; focused reruns and +nested subprocess summaries are excluded from that count. Build, formatting, +diff checks and all 96 Python harness tests pass. The driver preserves library +exit 101. Linux-only fork cases were not executed locally. +Commands, source fingerprint and log digests are in +[browser evidence](evidence/serving-certified-browser-20261004.json). +The frozen inventory has 482 source/schema/manifest files, +including 468 Rust files. Exact SDK pins and protected original +index/archive remain unchanged. + +Next priorities are certified cache hydration and the five remaining legacy +reader failures, followed by owned HTTP/SSH/generated producers and actual +owner-aware remote routes. Complete-history native workspaces, efficient bounded +batch reads/cold I/O, physical owner adoption, custody archival, final DDL, +GC/backup/restore, OS containment, native maintenance, signed completion, +attribution and full-history/10,000-SDE capacity qualification remain mandatory. + +## Certified native object body checkpoint + +`ServingSnapshot::body` now resolves an object through the accepted catalog and +verified preferred source. It returns absent for entries missing from that +catalog even if a reused pack physically contains the object. Foreground copies +are bounded to 64 MiB and must match canonical kind, size, Git OID and BLAKE3 +body digest; incomplete or corrupt batch frames poison the reader. + +Resident file services share authenticated, verified native pack copies across +generations, coalesce cold misses with sixteen fixed stripes, and charge at most +eight live file slots with four cached copies. Borrowed/evicted files and native +processes retain slots. Idle cache eviction can relieve shared disk pressure; +retention follows deferred/failed cleanup. Native work uses the node's shared +foreground scope. Physical generation guards survive blocking creation, transfer, +verification and hashing, plus process descendants/reaping after cancellation. +Idle cache files retain file/root charges, not generation pins. + +The [serving contract](design/certified-serving-pins.md#certified-bounded-native-body-reads) +defines these bounds. Actual pack fixtures cover both formats, parallel reuse, +metadata/body integrity, disk-pressure eviction, live slot exhaustion, generation +isolation, current access after cached/in-flight reads, canceled provider work, +and process ownership. Their catalog installation is trusted fixture injection +to isolate serving behavior; this is not producer/publication qualification. +All 97 focused serving/cache/native families pass in 26.35 seconds on the final +source. Workspace/all-target Clippy with warnings denied passes in 38.39 seconds. +The workspace library runs 689 unique cases: 684 pass and the same five legacy +`objects` readers fail. All 385 publication cases pass. Nine selected portable +workspace/lifecycle cases pass, giving 698 unique Rust cases, 693 passes and five +failures; nested subprocess summaries and focused reruns are not counted twice. +The server build, formatting, diff checks and all 96 Python harness tests pass. +The driver retains the actual library exit 101. Linux-only fork cases were not +executed locally. Commands, source fingerprint and log digests are retained in +[native body evidence](evidence/serving-native-bodies-20261004.json). The frozen +inventory contains 478 source/schema/manifest files, including 464 Rust files. +Exact SDK pins and protected original index/archive remain unchanged. + +At this historical body checkpoint, browser/tree/file/history/graph consumers +and remote routes still required conversion; the newer browser checkpoint above +converts local object and comparison ancestry consumers. Streaming larger bodies, a complete graph workspace for +native history, persistent batch scheduling and cold-pack I/O optimization remain +open. One private source pack is not necessarily a complete Git history workspace. +Full producer/reader conversion, physical owner adoption, custody archival, +final DDL, GC/backup/restore, OS containment, native maintenance, signed completion, +attribution and full-history/large-team capacity qualification remain mandatory. + +## Certified browser ref reads checkpoint + +The local browser's ref listing and resolution now select the accepted immutable +joint ref root through `ServingSnapshot`; neither legacy ref rows nor their +default branch/counter can override it. The implementation reuses `RefPage`, +`RefExpectation` and the existing byte-ordered `RefStateIndex`. Its bounded node +client is shared by the resident, and each pin coalesces ref-descriptor loading. +Ref and metadata reads share admitted private work, current access/owner/lease +checks before and after I/O, and physical-drain ownership across observer loss. + +Pages contain at most 256 records and 512 KiB of name/record charges. Live cursors +skip deleted subtrees; version-preserving consumers can include tombstones. +Continuations bind the ref snapshot's generation and return changed on mismatch. +Old borrowed generations remain immutable while new heads are selected. HTTP +requests remain bounded at 512 KiB, allowing legal long cursors with JSON escapes; +overlong names reject as invalid rather than exceeding the index's byte limit. +The [serving contract](design/certified-serving-pins.md#certified-immutable-ref-reads) +defines these boundaries. + +Five new ref families and one actual browser HTTP/manager family exercise both +formats. They cover count/byte continuation, retained deletion versions, old-head +immutability, cached and in-flight access revocation, blocked real provider I/O +after canceled observation, bad snapshot context, and deliberately conflicting +legacy SQL. All 62 focused serving/recovery families pass in 20.59 seconds; +warnings-denied workspace/all-target Clippy passes in 28.37 seconds. Inventory +roots in the ref fixtures are trusted injection to isolate reader behavior, +not proof of native graph publication or capacity. + +Final frozen-source library qualification executes 679 unique cases: 674 pass +and the same five legacy `objects` readers fail. All 381 publication, seven +startup, four resident recovery and four resident serving cases pass. Nine +selected portable workspace/lifecycle cases pass: 688 unique Rust cases, 683 +passed and five failed. Nested subprocess summaries and focused cases are not +counted twice; the driver's actual library exit 101 remains visible. Linux-only +fork cases were not run locally. Source fingerprints, commands, log digests and +corrected fixture diagnostics are retained in +[certified ref evidence](evidence/serving-refs-20261004.json). +The server build, formatting, diff checks and all 96 Python harness tests pass. +Static qualification verifies 474 frozen source/schema/manifest files, including +460 Rust files, exact SDK pins and the protected original index/archive. + +Highest next work is certified object/body/native serving and owner-aware remote +routes, followed by conversion of all remaining ref/cache/graph/browser/policy/ +merge consumers and HTTP/SSH/generated producers. Default-branch and other +policy producers still require immutable publication conversion. Physical +adoption/quota recovery, admitted immutable custody history/exact lookup, final +DDL, typed GC/backup/isolated restore, fault campaigns, OS containment, native +acceleration/rewrite/fair maintenance, signed completion/cold clone, attribution +and full-history/10,000-developer capacity qualification remain mandatory. +The full objective remains open and the branch remains unreleasable. + +## Serving construction/shutdown barrier checkpoint + +A deterministic regression exposed a resident constructor publishing its weak +serving capability before registering its lifecycle owner in the manager's drain +inventory. Shutdown could collect that inventory while construction was pending. +Registration now precedes capability exposure under the loaded-resident mutex; +shutdown permanently closes construction under the same mutex before collecting +pools. A rejected constructor joins its private pool, scanners and exact recovery +inside its existing tracked residency task before returning `CellDraining`. +Workspace, Cell and publication ownership therefore outlive that cleanup. + +The regression fails against the preceding publication ordering with +`unpublished constructor exposed serving before registered ownership`; the fixed +case passes in both formats. All 56 focused serving/recovery families pass in +19.40 seconds. Warnings-denied workspace/all-target Clippy passes in 31.41 seconds. +Final frozen-source library qualification executes 673 unique cases: 668 pass +and the same five legacy `objects` readers fail. All 376 publication, seven +startup, four resident recovery and three resident serving cases pass. Nine +selected portable workspace/lifecycle cases pass, giving 682 unique Rust cases, +677 passed and five failed. Nested subprocess summaries and focused cases are +not counted twice. This remains an incomplete release gate, with the library's +actual exit 101 retained. Linux-only fork cases were not run locally. +The server build, formatting, diff checks and all 96 Python harness tests pass. +Static qualification verifies the frozen source, exact SDK pins, protected +original index/archive and local documentation links. + +The source fingerprint, commands, log digests and original failing regression +are retained in [shutdown barrier evidence](evidence/serving-pool-shutdown-20261004.json). +These lifecycle fixtures do not qualify full nonempty native/body/history serving, +physical owner adoption or large-team capacity. Highest next work remains the +authoritative reader and producer conversion listed below; all remaining release +requirements stay open. + +## Resident serving pool checkpoint + +The production manager now owns one node serving budget and creates one lazy, +repository-scoped pool/context with shared index/file clients for each local +resident. `RepositoryCell::serving_snapshot` uses a weak association. The pool +coalesces pending viewers, retains at most four generation slots including +closing owners, and binds returned snapshots to the actual accepted joint fact +across head races. At capacity it retires an idle old owner, returns explicit +backpressure and charges the slot until real producer completion. + +Eviction pauses borrowing and producer creation, refuses pending originals/ +borrows/physical workers without discarding them, and resumes the same owners +on refusal. Idle owners release through the existing exact-token admission gate +before queue closure. A private task owns accepted drain across observer +cancellation. Pool cleanup and generation releases proceed independently. +Shutdown joins serving owners before closing the node tracker/publication +budget or Cell/workspace/heartbeat. Partial construction joins its first owners. +See the [serving contract](design/certified-serving-pins.md). + +Fifty-five focused serving/resident families pass in 20.42 seconds; they include +six new pool and two actual-manager cases in both formats. The cases cover +coalescing/current access, canceled cold observation, lost acknowledgement, +four-generation retention and real slot reuse, deterministic acquisition head +races, busy/canceled exclusive drain, independent release while an old real +provider blocks, weak repository association, eviction retry and actual server +shutdown held by the last borrowed clone. Warnings-denied workspace/all-target +Clippy passes in 28.89 seconds. Empty/copied-root fixtures qualify this lifecycle, +not full nonempty native/body/history serving or capacity. + +Final frozen-source workspace library qualification executes 672 unique cases: +**667 pass and five fail**, retaining exit 101. All 376 publication, seven +startup, four resident recovery and two resident serving families pass. Two +nested subprocess summaries are excluded and focused cases are not counted +again. Nine selected portable workspace/lifecycle cases also pass; combined +coverage is 681 unique cases, 676 pass and five fail. No compatibility table or +green-result substitution is introduced. The server build passes in 32.16 +seconds and formatting in 1.35 seconds. All 96 Python harness tests pass. +The freeze covers 471 Rust/SQL/manifest files, including 457 Rust files. The +[persisted evidence](evidence/serving-pool-20261004.json) records commands, source +fingerprint and log digests. The initial check's missing Arc qualification is +retained as a draft diagnostic, not a passing check. The final driver completes +its other checks and exits 101. Both preceding-head Linux Rust CI failures have +been read and contain exactly the same five `objects` readers; they qualify the +preceding source only. Linux-only fork integration cases were not run locally. + +Highest next work is conversion of all authoritative object/ref/cache/graph/ +browser/policy/merge readers and HTTP/SSH/generated producers to the resident +snapshot and certified publication path, including physical native/body/stream +ownership. Five unconverted legacy `objects` readers still fail. Physically +fenced adoption/quota recovery, immutable admitted custody history/exact lookup, +final DDL, typed GC/backup/restore, fault campaigns, OS containment, native +acceleration/rewrite/fair maintenance, signed completion/cold clone, attribution +and full-history plus 10,000-developer capacity qualification remain mandatory. +The full objective remains open and this branch remains unreleasable. + +## Owned serving lifecycle checkpoint + +`ServingOwner` now owns acquisition, accepted-original physical handoff, +automatic renewal and exact release independently of request observers. Its +private restart supervisor retains factory identities, original commands and +held/uncertain tickets. Snapshot clones share a private borrow lifetime; +closure refuses new borrows and renews existing ones until their last guard +drops. Physical provider work drains before release, including work detached by +observer cancellation. A known denied release retains ownership and permits a +new proof only after the denied original has settled. Read revocation after a +lost acquisition acknowledgement still retains and authentically drains the +accepted pin. No lease expiry, timeout or logical denial deletes physical roots. + +The accepted handoff authenticates the exact original acquisition ordinal and +current retained pin/generation under actual owner checks. Restored commands, +renewals, absent/denied executions and independent duplicate physical owners +cannot mint that handoff. Owner, snapshot and physical-I/O budgets are separate, +bounded node/account shares; a snapshot cannot consume every I/O slot. The new +production files participate in the RepositoryModule code digest. See the +[serving contract](design/certified-serving-pins.md). + +Thirteen focused families pass in 8.29 seconds after compilation, including six +transport fault modes, five producer restart points, clone renewal, deterministic +recovery/revocation, canceled release observation and blocked actual provider +I/O. Their initialized empty catalogs qualify lifecycle ownership, not full +nonempty native/body/history serving or capacity. + +Final frozen-source library qualification executes 664 unique cases: **659 pass +and five fail**, with exit 101 retained. All 370 publication, seven startup and +four resident-recovery families pass. Two nested subprocess summaries are +excluded and focused cases are not counted twice. Nine selected portable +workspace/lifecycle cases also pass, including canceled prebound startup. +Combined coverage is 673 unique cases, 668 pass and five fail. Every failure is +one of the five unconverted legacy `objects` readers. Warnings-denied +workspace/all-target Clippy passes in 29.19 seconds, the server build in 37.26 +seconds, formatting in 1.15 seconds, and all 96 Python harness tests pass. +The freeze includes 468 Rust/SQL/manifest files, including 454 Rust files and +the design SQL fixture. The [persisted evidence](evidence/serving-owner-20261004.json) +records the actual failures, commands and source fingerprint. Library failure +remains exit 101; the validation driver also finishes 101 after completing its +other checks. Initial driver prefix/platform assertion mistakes are recorded +separately, and successful completed commands are reused rather than rerun. +Linux-only fork cases were not executed locally; their latest prior CI does not +qualify this new source. + +Highest next priorities are the bounded resident generation pool, shared +node budgets and repository-scoped contexts, eviction/shutdown joining before publication-budget and +Cell/workspace closure, and conversion of every authoritative reader and +HTTP/SSH/generated producer. Physical fencing/adoption and quota recovery, +admitted immutable custody history/exact lookup, final DDL, typed collection, +backup/restore, fault campaigns, OS containment, native acceleration/physical +rewrite/fair maintenance, signed completion/cold clone, file attribution and +full-history plus 10,000-developer capacity qualification remain mandatory. +This checkpoint does not complete the full goal or make the branch releasable. + +## Resident recovery lifecycle checkpoint + +The production repository manager now retains one recovery coordinator plus root +and custody scanners for each initialized local resident, sharing node command +and read-round budgets. Scans are tracked by the existing server task tracker. +They pause without abandoning an owned query/artifact read and resume on the same +cursor when a busy coordinator refuses eviction. Idle coordinators close under +their admission lock; scans and Git maintenance join before Cell release and +workspace deletion. Runtime idle generation is refreshed after joining reads. +Release errors retain the runtime-handle refresh path. Remote routes have no +local recovery scanner. Partial worker startup explicitly joins its first worker. +See the [resident contract](design/resident-publication-recovery.md). + +Shutdown owns bounded concurrent per-repository drains, so one producer-held +command cannot delay another repository's exact recovery. Uncertain tickets keep +their original identity, command, receipt and admission. Held proofs stay owned +by their producer. Neither an observer timeout nor budget closure permits early +Cell shutdown, heartbeat withdrawal or workspace cleanup. The new scan, budget +and server lifecycle source files are included in RepositoryModule's code digest. + +The final-source library run executes 618 unique cases: **613 pass and five +fail**, with exit 101 retained. All 324 publication cases, seven startup cases +and four real production recovery families pass; two nested subprocess summaries +are excluded. The failing set remains exactly the five unconverted legacy +`objects` readers. Warnings-denied workspace/all-target Clippy passes in 28.72 +seconds. Nine additional workspace/lifecycle cases pass in 3.62 seconds, +including cancelled prebound startup; their command takes 43.49 seconds with +compilation. Combined coverage is 627 unique cases executed, 622 pass and five +fail. Focused cases are not counted again. The server build passes in 29.38 +seconds, formatting in 1.14 seconds, and static/diff checks pass with 451 frozen +source/schema/manifest files including 439 Rust files, 148 local documentation +links, exact SDK pins and unchanged protected index/archive. Evidence is +`/tmp/canopy-resident-recovery-validation.json`. + +Retained draft diagnostics include sibling-module shutdown/target visibility +errors, the regression's immediate uncertainty observation, and Clippy's +`int_plus_one` rejection. The test uses its actual repository target and observes +the later terminal result without requesting recovery; the comparison now uses +`>` without addition/overflow. No diagnostic is treated as a passing run. No +compatibility table or green-result substitution is introduced. + +Full producer/reader/final-DDL conversion, admitted custody-history frames/exact +lookup, certified serving generation ownership, retained-input takeover/adoption, +scanner panic/restart and provider/owner-loss campaigns, typed GC/backup/isolated +restore, OS containment, native acceleration/physical rewrite/fair continuous +maintenance, signed completion/cold clone, file attribution and full-history plus +10,000-developer capacity qualification remain mandatory. This branch remains +local, unpublished and unreleasable. + +## Current-root selection and exact serving drain checkpoint + +Production registers bounded current-Read query 48 to observe the joint +catalog/ref head without allocating retention. The existing exact pin remains +immutable across head advances. Acquisition/renewal dispatch copies now share +their original encoded custody intent through an Arc, preserving identity while +avoiding duplicated command bodies. + +`ServingDrainAdmission` excludes new production work while admitting only a +bounded set of exact serving releases. Busy reservation changes no existing +admission. Closure requires every selected token's observed successful release +and a fully idle coordinator. Global close waits for that owner; dropping the +guard preserves admitted originals, recovery credits and sticky closure. A +reused reader ID with another admission sequence cannot satisfy the drain. +These are scheduling primitives; production generation pooling, ownership +handoff and residency/shutdown ordering remain unimplemented. + +Eleven focused families pass in 1.21 seconds. They cover both object formats, +current Read/public access and revocation, exact repository identity, joint head +selection with immutable older retention, indexed lookup, bounded framing, +busy/invalid reservations, exact-token exclusion, held dispatch, all three +release transport fault modes, lost observers, denied releases, global close, +guard cancellation and original-command recovery. The earlier fixture compile +failures are retained as diagnostics and are not passing qualification. + +Final frozen-source library qualification executes 651 unique cases: **646 pass +and five fail**, with exit 101 retained. All 357 publication, seven startup and +four production resident-recovery cases pass. Two nested subprocess summaries +are excluded; focused tests are not counted again. Nine additional workspace +and lifecycle tests pass in 3.44 seconds. Combined coverage is 660 unique cases, +655 pass and five fail. Every failure remains one of the five unconverted +legacy `objects` readers; no compatibility table or green-result substitution +was introduced. Warnings-denied workspace/all-target Clippy passes in 24.07 +seconds, the server build in 31.51 seconds and formatting in 1.08 seconds. +Static checks verify 463 frozen source/schema/manifest files, including +450 Rust files, 152 local doc links, exact SDK pins and unchanged protected +index/archive. Evidence uses `/tmp/canopy-serving-selection-drain-*`. The draft +diagnostic summary explicitly records the overwritten initial focused log; the +final frozen-source logs qualify this source. No capacity or complete production +reader claim follows from these results. + +## Durable serving command checkpoint + +Serving acquisition and renewal now use the existing durable custody journal and +publication queue through `ReadyServingCommand`. Raw 44/45 receivers are excluded +from production registration. Fresh command codecs 41/42/43 and authenticated +intent/stop domains use version 2; the journal key, indexed discovery and MACs +bind creating versus serving purpose. Registration requires Read for serving, +while acquisition allocates no creating namespace. The original logical account +and request digest persist through renewal. Exact known serving grants and +denials remain immutable knowledge rather than fresh physical read authority. + +Renewal retains a shared physical-drain guard through factory ownership, held +admission, dispatch and uncertainty. The queue reuses its foreground/account/node +budgets with the existing 28 KiB custody reservation; release/stop keep their +reserved maintenance class. Separate job kinds prevent collisions among serving +commands, serving stops, releases and creating requests with the same real ID. +The existing scanner visits both purposes and stops expired unexecuted originals +without deleting accepted serving roots. The [serving contract](design/certified-serving-pins.md) +details protocol bounds and the still-missing resident ownership handoff. + +Final-source macOS/Rust 1.98.0 qualification executes 640 unique workspace +library cases: **635 pass and five fail**, with exit 101 retained. All 346 +publication, seven startup and four production resident-recovery cases pass; +two nested subprocess summaries are excluded. The ten new serving integration +and codec families are included in that total. Nine additional workspace and +lifecycle cases pass in 3.60 seconds, including canceled prebound startup. +Combined: 649 unique executed, 644 pass and five fail. The failure set remains +exactly the five legacy `objects` readers awaiting certified-root conversion. + +The new families cover both formats, all six registration/execution transport +fault modes, canceled and closed observation, held discard, atomic late/ignored +phase rollback, recorded revocation, actual cold owner restoration after SDK +expiry, same-ID purpose separation, page-one scanner traversal, immutable +identity/shared pending quota, framing, role mismatch and v2-only decoding. The +first eight-family run passed in 3.32 seconds. Preliminary enum-size, moved-test- +guard and invalid UUID fixture diagnostics remain under the draft/pre-UUID log +prefixes; they are not passing qualification. No compatibility fallback or lint +suppression was introduced. + +Warnings-denied workspace/all-target Clippy passes in 25.42 seconds, the server +build in 29.78 seconds and formatting in 1.07 seconds. Static/diff checks verify +461 frozen source/schema/manifest files including 448 Rust files, 152 local doc +links, the five SDK manifest/six lock pins, a clean SDK and unchanged protected +original index/archive. Proof and final-source logs use the +`/tmp/canopy-serving-custody-*` prefix. The qualification driver terminates with +zero only after explicitly recording the failed library run and passing the +remaining checks; this is not a green workspace test result. + +Production resident acquisition/renewal ownership and bounded generation pooling, +physical handoff before detached observers, all object/ref/graph/native/stream +consumer conversion, old-owner physical fencing/adoption/quota recovery and +admitted immutable custody history with exact lookup remain immediate priorities. +All full producer/final-DDL, typed GC/backup/isolated restore, OS containment, +native acceleration/physical rewrite/fair maintenance, signed completion/cold +clone, file attribution and full-history/10,000-SDE capacity gates remain open. +This local branch remains unpublished and unreleasable. + +## Certified serving pin foundation + +The [serving contract](design/certified-serving-pins.md) describes the new atomic +Read-only generation receiver, separate bounded retention table, owner-checked +metadata capability and tracked physical-drain/release protocol. It is not yet +connected to production acquisition/renewal, object/native/stream consumers or +snapshot caching. Expired serving pins remain retained until authenticated drain; +no expiry/epoch-only cleanup is introduced. The original goal remains active. +Final-source macOS/Rust 1.98.0 qualification passes all twelve focused serving +families (1.63 seconds), including independent-context duplicate exclusion, +blocked-provider cancellation/revocation, real owner restoration, and original +release absence/lost-reply/panic in both object formats. Initialization's recovery +floor is retired through its authentic terminal release before checking serving +reaping. A separate typed scheduler key prevents logical-ID collisions with held +preparation jobs while retaining the shared node/class/account budgets. + +The full workspace library remains **failed** (exit 101): 625 pass and the same +five unconverted `objects` readers fail, out of 630 unique cases. All 336 +publication, seven startup and four production resident-recovery cases pass +within that run; two nested subprocess summaries are excluded. Nine additional +workspace/lifecycle cases pass in 4.42 seconds, including prebound cancellation. +Combined: 639 unique executed, 634 passed, five failed. Clippy workspace/all-targets +with warnings denied (23.28 seconds), server build (30.96 seconds), formatting +(1.08 seconds), diff/static checks, 459 frozen source/schema/manifest files +(446 Rust), 151 local doc links, exact five SDK manifest/six lock pins and the +protected original index/archive checks pass. Retained proof and logs use the +`/tmp/canopy-serving-pins-*` prefix. Draft diagnostics are retained, not counted +as passing qualification. + +The bounded process-wide owner registry closes the duplicate-drain-counter gap: +all contexts reject a second constructor for an owned exact pin, including with +an independent budget. It retains only weak entries, caps live owners at 4,096, +and holds no async/provider work under its lock. This local physical exclusion +is not a durable acquisition ledger. Expired/old-owner SQL roots deliberately +remain retained until their actual ownership is resolved; automatic expiry or +new-epoch cleanup would violate correctness. + +Highest next: a production serving owner must retain exact acquisition/renewal +commands, coalesce a bounded set of generation capabilities, and carry their +worker/stream lifetime through actual eviction/shutdown. Convert all actual +object/ref/cache/graph/browser/policy/check/merge consumers, including the five +failures. Complete owned HTTP/SSH/generated producers and final hard-cutover DDL; +admitted immutable custody history/exact lookup, retained physical-input +adoption and scanner/fault campaigns remain required. Typed GC/backup/isolated +restore, OS containment, native acceleration/physical rewrite/fair maintenance, +signed completion/cold clone, file-attribution endpoint/UI/cache/index, and full +Linux/Kubernetes/Chromium plus 10,000-engineer mixed-load/recovery/capacity gates +remain mandatory. The branch is local, unpublished and unreleasable; no capacity +claim or whole-goal completion is made. + +## Shared node publication budget + +Repository dispatchers now require an explicit `PublicationBudget`, reusing the +existing limits profile, private ready values and exact SDK command ownership. +Clones share operation, command-byte and per-account reservations across +repositories. Foreground and maintenance retain independent shares; maintenance +account admission leaves room for another account. Separate account and class +transport gates are acquired before copying the dispatch body. Account waiters +cannot consume the node slots needed by another account. Uncertainty releases +transport capacity but retains command credits; a known result or proven held +discard drops retained resources before returning credits. Closing the node +budget rejects new admission and preserves activation/recovery of originals +already admitted. The repository queues retain their FIFO/account rotation and +class-burst behavior. See the [dispatch contract](design/shared-publication-dispatch.md#shared-node-admission-and-transport). + +Six new regression families cover aggregate account/byte/class admission, +transport headroom and FIFO progress, partial-gate cancellation, invalid/overflow +bounds, exact returned ready identities, and actual cross-repository dispatch in +both object formats. Existing absent/lost-reply/panic recovery now also checks +that the node budget remains charged through closure and returns credits only +after exact resolution. The native held-discard regression checks node credit +return after dropping the verified proof. All 82 constructor call sites now +supply a budget. The first draft had missing exports and three fixture ownership +references; the first focused run had a test awaiting a later gate waiter before +its existing FIFO predecessor. Those diagnostics are retained; the fixture now +observes the real FIFO order without changing the implementation or capacity. -Implementation is isolated in the PR worktree. The original checkout contains an unrelated, extensive staged workspace merge; its workspace, benchmark and runtime work has been preserved. Canopy is split into Git-format, object-storage and server crates. Main now contains all completed PR #20–#30 changes through [PR #31](https://github.com/crabbuild/canopy/pull/31), merged at `db80fd836db94fff894030f02d736fe92840748c`. The audit verifies each directly merged PR's exact merge tree and main ancestry; the entire main tree is identical to completed PR #30 (`5bf48677857e3d1dd769aa7f1d73eb5db00db30f`). PRs #28–#30 originally merged into stack branches and reached main through #31. Both #31 Verify runs, [37132349361](https://github.com/crabbuild/canopy/actions/runs/37132349361) and [37132329706](https://github.com/crabbuild/canopy/actions/runs/37132329706), pass harness and Rust. The merged main revision also passes [Verify 37132672371](https://github.com/crabbuild/canopy/actions/runs/37132672371). +Final-source macOS/Rust 1.98.0 qualification passes the focused eight-test run +in 0.91 seconds and all 320 publication plus seven startup cases within the +full workspace library. The library remains **failed** (exit 101): 605 pass and +the same five unconverted `objects` readers fail, out of 610 unique cases; +two nested subprocess summaries are excluded. Nine additional real +workspace/lifecycle cases pass in 3.36 seconds, including prebound startup. +The combined result is 619 unique cases executed, 614 pass and five fail; +focused cases are not counted twice. Warnings-denied workspace/all-target +Clippy (25.83 seconds), the server build (32.10 seconds), formatting/diff, +447 unchanged source/schema/manifest hashes including 435 Rust files, +141 local documentation links, exact SDK pins and protected index/archive +checks pass. Evidence is `/tmp/canopy-node-publication-validation.json`. -All five Cellule dependency declarations and six lockfile entries pin `161067f5a21703b3e257024bcb64e565fd9657b4` from [Cellule PR #50](https://github.com/crabbuild/cellule/pull/50), including the admitted owner fence, exact-command snapshot and admitted-mutation APIs. Historical validation below remains attributed to its original source revisions. Trusted ref-plan certification, typed catalog/ref publication, immutable exact-response completion and the class/account-fair dispatcher exist. Production startup/HTTP/SSH/generated producer and reader conversion, mandatory registration and the fresh-schema hard cutover remain open. +At the preceding `4d67294` checkpoint this was an admission primitive without +production wiring. The resident recovery increment above now supplies the shared +owner, read-round admission and release/drain lifecycle. Integration of every +remaining production consumer and provider/resource qualification remain open. The wire +credits do not qualify whole-process heap/RSS, native descendants, provider +traffic or capacity. The full producer/reader/final-schema cutover, admitted +history archival/exact lookup, retained-input adoption, certified serving +ownership, typed GC/backup/isolated restore, OS containment, continuous fair +maintenance/native acceleration/physical rewrite, signed completion/cold clone, +file attribution and complete Linux/Kubernetes/Chromium/10,000-developer mixed +load remain mandatory. This local checkpoint remains unpublished and +unreleasable. + +## Automatic staging retirement in progress + +The local staging coordinator now owns one bounded read-only probe over its admitted uncertain custody commands. Exact ordinal plus authenticated intent fingerprint finds an old stopped original even after a successor becomes the latest head. Only authenticated logical stop schedules existing exact recovery; absent commands, unavailable/corrupt private metadata and known execution phases alone retain their original reservations. Checkpoint/final commands keep separate owners. The existing fence/drain path joins running callbacks and drops retained resources before returning worker and operation admission. Observer drop and service closure do not discard this ownership. Probe failures/restarts/recovery scheduling are visible in bounded service counters. See the [lifecycle contract](design/staging-service-lifecycle.md#automatic-observation-of-custody-retirement). + +The original regression timed out after a genuine stop because manual staging recovery was still required (`/tmp/canopy-stage-stop-red.log`). The first draft compile caught a changed helper signature used by publication observation; the signature is preserved and probing starts only for custody uncertainty. The corrected regression passes. Seven added regression families now pass, including all seven custody actions in both formats, closed-service observer loss, staging/bound callbacks and retained-result drop ordering, private-query failure and known-phase non-retry, exact old-ordinal lookup after a successor, malformed facts and bad-head repair/fair progress, checkpoint exclusion, and 130 admitted operations over multiple pages plus restart at an earlier key after idle. Initial extended fixture failures are retained: the warm test observed the previous binding; the checkpoint test assumed a nonexistent result accessor and tried the fresh-operation factory against an existing journal; the page test used a forbidden all-zero operation ID. The fixtures now wait for the actual uncertain renewal, use the existing checkpoint wait API and an explicit registered successor, and use valid nonzero keys without changing production guards. + +Final-source macOS/Rust 1.98.0 checks pass the seven-family focused run in 46.08 seconds and all 314 publication plus seven startup cases within the full library run. The full library remains **failed** (exit 101): 599 pass and the same five unconverted legacy readers fail, out of 604 unique cases; two nested subprocess results are excluded. Nine real workspace/lifecycle cases pass in 3.35 seconds, including the prebound-listener regression. This is 613 unique Rust cases executed, 608 pass and five fail; focused cases are not counted twice. Workspace/all-target Clippy with warnings denied (24.38 seconds), server build (29.14 seconds), formatting/diff, 444 frozen source/schema/manifest files including 432 Rust files, 140 local documentation links, exact SDK pins and protected index/archive checks pass. Evidence is `/tmp/canopy-stage-stop-validation.json`. This is not production HTTP/SSH producer conversion, general repository scanner lifecycle wiring, whole-process memory/I/O qualification or large-team capacity. Those gates and the five known unconverted legacy-reader failures remain open. + +## Owned staging and preparation conversion in progress + +The local staging service now prepares both the exact custody original and its exact registrar for all seven custody actions. The fair preparation dispatcher uses the same owner for Claim/Renew, including original-head reconstruction. Direct caller-owned raw session/base renewal APIs have been removed; base renewals prepare a ready command for transfer to the service. Jobs retain both originals through uncertainty, observer cancellation and closure. Recorded phase knowledge precedes SDK expiry and local guards. Restored renewals on an existing session/base retain the original shared fence; a failed fresh query fences all of its existing resolvers. New execution after authoritative absence checks the local fence, deadline and residence ceiling. + +Custody reservation is 28 KiB; an admitted staging checkpoint raises it to 32 KiB. Existing operation, worker, actor, class and global byte caps have not increased. Twenty-seven staging service tests pass, including original reply loss/expiry, renewal/revocation, actual owner Claim, registrar loss before submission/after acceptance/panic, canceled observers and closed-service recovery in both formats. At checkpoint `1a11162`, final-source macOS/Rust 1.98.0 checks passed: 286 publication cases in 214.01 seconds and nine real startup/workspace lifecycle cases in 5.17 seconds, totaling 295 unique focused Rust tests. All-target workspace Clippy passes with warnings denied, the server binary builds, formatting/diff checks pass, and 423 Rust source hashes plus 90 local documentation links and the unchanged protected checkout/SDK pins are verified. The proof is `/tmp/canopy-registered-services-validation.json`. These checks qualify this local service checkpoint, not the whole production cutover or team capacity. + +Cold session construction now carries mandatory `PreparationAuthority`, bound to the exact Cell target and backed in production by startup's validated Cell Control/live node advertisement. It checks actual incarnation/epoch before and after the lease query; staging probes, bound handoff, base catalog/frontier loading and standalone positive recovery share this source. An owner observation failure permanently fences existing shared sessions, while original known outcomes remain recoverable. New registered Claim can restore current-owner custody. The regression set includes prior-owner cold restore, genuine current-owner takeover and missing/corrupt Control records in both formats. Final-source macOS/Rust 1.98.0 qualification passes 289 publication cases in 176.00 seconds and nine real startup/workspace lifecycle cases in 3.61 seconds, totaling 298 unique focused Rust tests. All-target workspace Clippy with warnings denied, the server build, formatting/diff checks, 424 frozen Rust-source hashes, 134 local documentation links, the five manifest/six lockfile SDK pins and protected index/archive checks pass. Evidence is `/tmp/canopy-owner-validation.json`; the original failing regression is retained in `/tmp/canopy-cold-owner-red.log`. These checks do not establish the full hard cutover or large-team capacity. The following cold-staging increment reconstructs registered custody heads and qualifies shared-fence callback drain. The next custody priorities are bounded unresolved-head stop/scan, settled-history archival and production takeover/input adoption. The full producer/reader/final-schema cutover, serving retention, typed collection/backup/restore, resource containment, maintenance/acceleration and full-history/team capacity gates remain open. + +Cold staging now reconstructs the latest authentic registered custody head through `ReadyStaging::restore` for every staging/preparation Begin/Claim/Renew and Bind. Original evidence and positive/negative receipts remain observable before fresh owner/lease probes and after fencing or shutdown. Native work requires fresh custody; old clocks cannot open sessions. Absent old-token Renew/Bind settle their exact original stale denial under a new owner; expired unresolved originals remain uncertain and charged instead of being replaced. The shared session now signals its permanent fence to in-flight bound callbacks and late subscribers. Cancellation joins callbacks and drops resources before returning their existing worker credits. Seven SHA-1/SHA-256 regression families pass in the focused run (17.88 seconds); the real owner-loss worker test failed before the signal was added (`/tmp/canopy-cold-staging-worker-red-fixed.log`). Final-source macOS/Rust 1.98.0 qualification passes 296 publication tests in 181.99 seconds plus nine real startup/workspace lifecycle tests in 3.16 seconds: 305 unique focused Rust tests. Workspace/all-target Clippy with warnings denied, the server build, formatting/diff checks, 426 frozen Rust-source hashes, 134 local documentation links, the five manifest/six lockfile SDK pins and unchanged protected index/archive checks pass. Evidence is `/tmp/canopy-cold-staging-validation.json`. These results qualify the local recovery/callback checkpoint, not the full cutover or large-team capacity. The 28 KiB custody/32 KiB checkpoint command-wire reservations and existing operation/actor/worker caps are unchanged; whole-process resident use and native OS containment are not qualified. This checkpoint does not reconstruct input inventories or production producers/readers. Bounded unresolved-head stop/scan and settled-history archival remain the immediate custody priorities. + +The proposed [directory file attribution](design/file-attribution.md) uses commit-pinned asynchronous page batches and bounded history caching, with an optional shared immutable index. Its native experiment passes 101 path comparisons across 17 commit states. Its production endpoint, UI, cache/index and load qualification remain unimplemented. + +## Bounded expired-custody retirement in progress + +The local cutover now has a separate authenticated stop record for expired unresolved custody originals. It preserves the original command identity, bytes, SDK expiry and absence of an execution result while freeing pending-head quota; it grants no native work, generation retention or remote deletion. Command 43 rechecks the exact intent, receiver expiry and actual owner atomically. Known original results and first-writer stops remain immutable. An explicit successor can advance from logical closure without rewriting old history. The typed stop outcome distinguishes the original custody evidence, retirement invocation, first-writer fact and any genuinely known invocation receipt. + +`CustodySupervisor` reuses bounded recovery limits and seeks only operation keys through the existing partial pending index. It advances past corrupt heads and wraps for new/failed keys, using the existing account-fair maintenance dispatcher rather than a second outbox. Existing maintenance slots/worker shares and the 8 KiB wire reservation are unchanged. Exact uncertain retirement commands are recovered even when their accepted marker removes the key from discovery. Existing staging recovery observes a typed stop and drains without a fabricated original reply. Eleven focused families pass (19.77 seconds), including authenticated-record transplant rejection and retirement alongside its own pending preparation. The dispatcher uses a distinct retirement key kind with the same real operation ID; it resumes only a preparation whose exact original evidence matches the stop, fencing that shared session before releasing admission. Final-source macOS/Rust 1.98.0 qualification passes all 307 publication tests (191.91 seconds), nine real startup/workspace lifecycle tests (3.24 seconds), warnings-denied workspace/all-target Clippy, the server build and formatting. The full workspace library run passes 585 cases and fails five (590 unique cases; nested subprocess runs excluded): four legacy object-read tests and one fetch-reachability test query the removed `objects` table. That table was already absent from the preceding checkpoint schema; these consumers still require the planned authoritative-reader conversion. The failure is retained in `/tmp/canopy-custody-stop-workspace-library-final.log`; no legacy schema fallback has been added. The 307 focused publication cases are contained in the library run and are not added to its total. Static qualification verifies 441 frozen source/schema/manifest files including 429 Rust files, 135 local documentation links, the exact SDK pins and the unchanged protected index/archive. Evidence is `/tmp/canopy-custody-stop-validation.json`. Production lifecycle wiring and automatic staging/startup resumption outside the preparation dispatcher, settled-history archival and removal of redundant first-admission columns remain required. The full producer/reader/schema, retention/GC/backup/restore, OS containment, maintenance/acceleration, attribution and full-history/team capacity gates remain open. + +## Production startup retirement in progress + +The production initializer now recognizes authenticated stopped originals and chooses an explicit successor from a receipt-watermarked indexed observation of the current operation. It distinguishes a wholly absent operation from inconsistent repository identity, actor, digest, generation or token metadata. Its existing tracked/account-bounded transition can retire its own expired unresolved head through the shared private stop factory/exact completion; known original grants/denials still take precedence over SDK expiry. This does not instantiate general resident-repository discovery or automatic staging recovery. No new native authority comes from retirement, and no compatibility schema or synthetic original result is introduced. + +The original regression returns `InvocationError::Pending` after an authentic manual stop against the preceding initializer (`/tmp/canopy-startup-stop-red-fixed-fixture.log`). The first composed four-family run passes in 11.77 seconds. The subsequent expanded run passes six and fails one fixture setup because its injected SQL edit reserved zero mailbox bytes; the fixture now charges its actual SQL bytes using the existing SDK admission contract. Final-source macOS/Rust 1.98.0 qualification passes all seven startup families in 16.12 seconds, including automatic retirement after real owner restoration. The full workspace library run passes 592 cases and fails only the same five unconverted readers (597 unique library cases; two nested subprocess results excluded). All 307 publication and seven new startup cases are contained in that run and are not added again. Nine real startup/workspace lifecycle cases pass in 3.48 seconds: 606 unique cases executed, 601 passed and five failed. Workspace/all-target Clippy with warnings denied (30.31 seconds), the server build (34.58 seconds), formatting, diff checks, 442 frozen source/schema/manifest files including 430 Rust files, 137 local documentation links, exact SDK pins and protected index/archive checks pass. The proof is `/tmp/canopy-startup-stop-validation.json`. General production scanner ownership/drain, stopped staging resumption, retained-input adoption and admitted history archival remain immediate custody work. The five legacy reader failures from the previous full library run still require the planned reader conversion. All remaining full producer/reader/final-schema, serving retention, typed GC/backup/isolated restore, OS containment, maintenance/acceleration, file attribution and full-history/team capacity gates remain open. + +## Durable custody command journal started locally + +The production cutover now registers exact custody metadata before an upload namespace exists. The journal reuses the SDK snapshot/body contract, authenticated carrier, `Stamp`, `Recorded`, domain admission logic and namespace/pin allocator. Command 41 first-writer registration and command 42 exact execution cover preparation/staging Begin, Claim and Renew plus Bind. Positive and denied results share domain writes and SDK acceptance atomically. Late errors or ignored SQL writes leave SDK resolution absent; exact retry preserves the original identity. Corrupt metadata, `Unknown` and `Expired` are never treated as absence. Historical grants do not grant current custody or become new generation-retention roots. See the [durable custody contract](design/durable-custody-command-intents.md). + +Actual repository startup now discovers its latest custody head before preparing another original identity, including lost Begin and successor Claim results. It requires the current owner and a fresh custody query before using a historical grant; known denied final attempts require explicit registered Claim. Indexed authenticated historical grants allow fresh allocation after successor reaping without restoring an old namespace/pin. Production unbinds raw custody commands 11–13, 24–26 and 28; their domain methods are reused inside command 42 and explicit qualification fixtures. The production registry now has 16 commands and nine queries, including separate expired-custody retirement command 43. No old custody contract is retained as a production fallback. + +The journal uses a bounded command relation rather than per-object metadata: 4 KiB intents, 1 KiB phases with 512-byte replies, at most one unresolved head per operation, 1,024 pending heads and a 65,535 ordinal ceiling. Primary/partial grant indexes bound discovery. Registering every custody transition currently adds a mutation before execution. This local append history remains conservatively retained; before release, compact settled metadata into immutable per-operation frames with exact historical lookup. Count actual commands and metadata growth in capacity gates. Do not publish this partial cutover merely because primitive tests pass. + +The preceding custody-journal checkpoint `dcef9c8` passed **283 publication tests** in 169.10 seconds and **nine workspace/lifecycle tests** in 3.58 seconds: **292 unique focused cases**, excluding repeated reruns. All-target workspace Clippy passes with warnings denied in 22.03 seconds; the server binary builds. Formatting/diff, 422 frozen Rust hashes, protected index/archive, clean SDK checkout and five-manifest/six-lock-entry SDK pins pass. Evidence is `/tmp/canopy-custody-intent-validation.json`; logs retain compilation failures, the deliberately rejected late writes and the initial query-plan failure. Twelve SHA-1/SHA-256 families cover original identities/receipts, pre-namespace persistence, denials, first-writer races, every transition, cold owner restore with deleted SQLite, actual SDK expiry, joint initialized bases, reaped successors, forgery/corruption, indexed lookup, late abort and silently ignored SQL writes. Real workspace tests check certified startup and byte-identical journal restore. The partial cutover still requires complete runtime/provider/Linux and capacity qualification. + +Highest priority remains production service conversion: StagingCoordinator, preparation/publication ready factories and direct session renewal must preserve these original intents under fair admission and process-loss reconstruction before their raw qualification bindings can be removed. Then close terminal history archival/retention, complete producer/reader conversion and final DDL removal, serving-generation ownership, typed collection/backup/restore, OS containment, continuous maintenance, physical rewrite/accelerated reads and full Linux/Kubernetes/Chromium histories with the 10,000-developer mixed-load gates. Unresolved expired identities remain protected; bounded orphan discovery and explicit lifecycle recovery remain open. No whole-goal or capacity claim is made. + +## First accepted preparation admission checkpoint + +The local cutover now retains the first accepted `BeginPreparation` in the existing logical request row, using the same private bounded admission record, MAC, `Stamp` and `Recorded` result as staging. The two receipt kinds have separate purposes, columns and validators. Preparation records the actual command sequence, which can differ from the attempt sequence when Begin observes an already-bound staging attempt. Immutable SQL guards retain the first result through later admission, Claim, completion and reaping. Receipt encoding or the final insert/update failing rolls back allocation, custody and SDK acceptance together. + +Pending initialization looks up this original admission before another Begin. Reuse requires the actual current owner fence and a fresh custody query at the original receipt; historical clocks grant no lease. Explicit Claim can authenticate and recreate a reaped original operation using a new namespace/pin under the executing owner and current catalog floor. It cannot displace an active successor or recreate a completed logical outcome. Admission-only rows are accepted by pristine initialization and matching compaction; current codecs are 11/2, 12/2, 22/2 and 31/3. Source hashing includes both shared and preparation-specific receipt implementations. See the [first preparation admission contract](design/initial-preparation-receipts.md). + +Seven new SHA-1/SHA-256 families pass focused qualification. They cover actual receipt sequences, an existing bound staging attempt, first-result immutability, insert/update rollback and exact retry, deleting local SQLite before fresh-owner restoration, real SDK expiry, reaping, forged tokens, completed-request refusal, revoked/expired custody, MAC corruption, purpose separation and no invented knowledge for denied or unexecuted Begin. The first broader audit finds old outcome-count assumptions and a late-failure trigger attached to INSERT rather than the current UPDATE. The fixture corrections count selected outcomes, preserve complete state hashing including both initial receipts, and require the actual injected error, SDK absence and exact original retry. The source-independent purpose test keeps scope/actor/operation/digest/MAC valid and varies only the purpose. + +Final frozen-source qualification passes all **271 publication tests** in 188.32 seconds and all **three actual production workspace/startup tests** in 3.97 seconds: **274 unique focused Rust cases**. Workspace/all-target Clippy passes with warnings denied in 105 seconds; the server binary builds in 34.71 seconds. Formatting/diff, 418 unchanged Rust source hashes, 41 checked local documentation links, protected original index/archive and five-manifest/six-lock-entry SDK pins pass. The original missing-receipt regression, pristine-state integration failures and subsequent fixture failures are retained in `/tmp/canopy-preparation-admission-*.log`. Proof is `/tmp/canopy-preparation-admission-validation.json`. These checks do not qualify the entire workspace runtime, provider compatibility, full histories or team capacity on this partial cutover. + +This remains an unpublished, unreleasable checkpoint. Denied Begin, competing raw Begin identities, pre-dispatch process loss and subsequent Claim/Renew still need durable original snapshots/results. A lost successor Claim is not attributed to the original Begin. Full production producer/reader conversion and final DDL removal, typed collection/backup/isolated restore, resource containment, continuous maintenance, accelerated reads/physical rewriting and full-history/10,000-developer capacity remain mandatory. Historical capacity results below do not qualify this increment. + +## Production hard cutover started locally + +The isolated `codex/packed-production-cutover` branch is aligned with merged PR #33 at `9438bb865959fb975d5349ba8b9908b461653821`. Its first production change wraps the existing `RootPurpose` at the unchanged `canopy-root-v1.json` key in the required `canopy-pack-v1` envelope. There is no legacy decoder. The root remains bounded to 4 KiB, including envelope overhead. Encoding borrows the original purpose and bounds source/pin strings before serialization. Reservations and completion still use the original conditional create/ETag CAS. + +Startup performs a read-only root/serving-purpose check before opening or reclaiming local state, storage probes, identity/release writes or Cell activation. It preserves the concurrent-initializer recheck when identity appears after the first marker read. Local workspaces use `canopy-pack-v1/` and the same format marker. An existing `runtime-v1` directory is rejected and retained; missing/unknown markers in the new directory are rejected before cleanup. Existing owner/worker exclusion and descendant fencing are preserved. Source/release hashing now includes the deployment format and workspace/startup code. All fixture and benchmark paths follow the new directory. + +Four new unit regressions fail against the original code. A real startup regression also fails because an unversioned deployment was admitted. After the change, all 16 deployment and nine workspace tests pass. The real startup and durable restart pair pass, including old/missing/unknown formats, completed backup and unfinished restore rejection before any workspace or identity writes. The final envelope-boundary regression passes in the full library audit: all 516 server library tests pass. All 96 Python qualification tests pass in 41.507 seconds. The first broad multi-server audit records 104 passed and one macOS `AddrInUse` failure in the late SSH publication-refusal family. The original error remains retained. That family passes alone; holding a competing listener in its released HTTP-port gap deterministically reproduces the same error. Retaining and handing off the bound listener refuses the competing bind and passes all three original refusal scenarios. The shared fixture and all three restore starts in that family now retain their listeners. Temporary diagnostics are removed. The uninstrumented broader integration rerun passes all 105 multi-server cases with four test threads in 419.98 seconds, including the original cancelled-startup and late SSH refusal cases. Owner restart, Repository Cell and Smart HTTP pass. Combined with the preceding unchanged-source library/CLI/contract runs, all 661 unique current workspace Rust cases pass; child-process summaries and focused reruns are excluded. Workspace/all-target Clippy passes with warnings denied in 45.68 seconds. All eight isolated RustFS compatibility cases pass on this same source, including the 4,096-ref mirror, SHA-256 native candidates, signed HTTP/SSH, filtered clones and SSH LFS. Workspace doctests finish successfully with no cases; the server binary builds in 26.91 seconds. The separate large-transfer gate still needs at least 40 GiB free on both scratch and provider volumes; complete histories and 10,000-developer capacity remain unqualified. Formatting/diff checks, all 409 frozen Rust-source hashes, three changed scripts’ syntax and 116 local documentation links pass. + +The next local increment selects `packs::publication::SCHEMA` through one `REPOSITORY_SCHEMA` constant for production RepositoryModule migrations, actual repository acquisition and maintenance recovery. Production and publication qualification share the same 20 command and nine query descriptors, deriving actual codec versions and bounded envelopes from the typed operations. Legacy Git commands 3–10 and inline publication/completion 18/19/query20 are excluded from production binding; the latter remain explicit qualification-only bindings until their callers and DDL are removed. Production creation now obtains an actual preparation lease, constructs the private empty-catalog proof and atomically publishes catalog/ref roots through command 31 before exposing the repository as Ready. Ready restoration verifies the exact retained initialization and authenticated immutable metadata without another Begin or artifact allocation. Only a known Stale/Expired Begin refusal allows an explicit Claim of the observed original attempt. Uncertain errors remain errors; they cannot select a fresh admission within that invocation. Initialization work remains owned by the existing tracked repository-transition task through HTTP cancellation and shutdown. + +The new real HTTP creation regression initially fails because production still selects the old `objects` table. Its bootstrap portion then passes for SHA-1/SHA-256. Extending it to fresh-disk restoration exposes a test read error: ordinary SQLite bypasses Cellule's authenticated sparse VFS and reads an incomplete placeholder. The restored HTTP load itself succeeds. The fixture now inspects the exact published immutable Cell root with Cellule's supported read-only reader. The missing-catalog fixture initially joins a nested key as one escaped path component, so it deletes a different key; using the same scoped provider and proving HEAD absence corrects that injection. Final missing-initial-catalog cold loads return exactly HTTP 503 without advancing catalog generation or artifact allocation. The lint audit also catches a synchronous read-only connection guard spanning awaits; lexical query scopes release it before artifact I/O. No production workaround or lint suppression is added. + +All 254 publication tests pass in 154.86 seconds, including the shared production-registry contract, using four threads and standard stacks. The final three workspace integration tests pass in 2.45 seconds, including both formats, fresh-disk restoration, missing retained metadata and the deployment/workspace exclusions. Workspace/all-target Clippy passes with warnings denied in 7.85 seconds. These are focused local checks. The broad full-workspace, provider and capacity results above remain attributed to their earlier source; unconverted production Git paths cannot be qualified by this increment. + +**This is local, unpublished work and is not a releasable packed deployment.** Production Git pushes, cache/object/ref readers and generated producers still call legacy storage APIs, whose tables and bindings are absent from the selected production contract. Their conversion, product graph/policy/check/review consumers, and deletion of temporary SQL refs and inline response/certificate/plan adapters must complete together before release. Final initialization now uses the same exact registered recovery protocol as policy/root publication; initial Begin/Claim/Renew and failure before registration remain incomplete. Typed initialization terminal retirement now releases eligible closed pins through the shared immutable receipt archive, preserving exact original receipts. Reconstruction of older orphan attempts still needs complete production background-service wiring. The new bootstrap is not proof of the complete recovery or retention protocol. Actual typed collection/backup/isolated restore, retained-input adoption/repreparation, OS containment, accelerated reads/rewrites, continuous fair maintenance and complete repository/team qualification remain required. Format checks, startup fixtures and primitive publication tests do not prove that wider completion. + +## Typed initialization retirement in the local cutover + +The terminal archive now reuses one immutable `catalog_recovery_receipts` row per original incarnation/admission sequence for both selected push completion and closed initialization. The original authenticated `Record`, `Bundle`, journal, predecessor frames and release receipt are unchanged in role; no durable queue or per-object ledger is added. Command 40 advances to codec 2 and purpose v2 without a compatibility decoder. SQL guards prohibit archive mutation, replacement and deletion, and require the exact archived certificate/phase before pin deletion. + +Positive retirement checks the immutable selected initialization, verifies its complete typed empty catalog/directory/ref graph, then inserts the archive and deletes only the exact original pin in the same SDK transaction. The immutable initialization fact stores its original pin key for indexed closed-attempt discovery. A known negative can retire only after its exact active binding closes; a successor of the same logical operation retains a separate namespace and original receipt. Unknown attempts remain pinned. The existing retiring scanner defers active negative initialization without allocating another SDK command and recovers uncertain release jobs even after their pin disappears. + +Actual repository startup retires the successful initial pin before serving, including recovered closed positive publication and a known denied original after successful Claim. Its existing tracked transition retains this bounded work through cancellation. Maintenance fencing is read from validated durable Cell Control with a live node advertisement, and the release receiver independently requires its actual admitted fence and current Admin. Both startup and retirement use the same typed empty-graph verifier. Fresh Ready restoration allocates neither a new preparation nor another artifact namespace. + +The original red regression fails with `Context` against the preceding code because completed initialization cannot retire. Six new families cover SHA-1/SHA-256, immutable archives, distinct old/new pin identity, actual Admin/fence rejection, rollback at the last delete, missing catalog/directory/ref metadata, known expiry after Claim and later positive publication, losing release identity, lost acknowledgement, saved-body loss, SDK expiry, real fresh-disk owner restore, and automatic recovery after pin disappearance. The first broad audit passes 261 and fails three existing live-owner scanner assertions: their unrelated completed initialization now legitimately retires. Their shared rooted setup now performs the same initialization retirement as production startup; push-retirement archive counts select the exact original pin rather than all archived operations. Original failure logs are retained. The first corrected setup covered only standalone policy fixtures; a second native setup required the same retirement. That audit then exposed 17 shared native assertions expecting a permanent second pin. An unchanged-source isolated reproduction confirms completed publication followed by a total-pin assertion failure (one actual pin versus two expected). Four native assertion helpers now select the original push incarnation/admission sequence and still require its active-operation and pin counts. Their custody checks retain only a copied token after dropping the original session; no ownership lifetime is extended. + +The complete frozen-source publication suite passes **264 tests in 175.94 seconds** with four threads and standard stacks, including all six new families and the original native uncertainty/retirement/policy-history cases. The last-write fixture requires the exact injected SQLite failure plus original SDK absence, so a prior gate refusal cannot satisfy the rollback assertion. Three actual production workspace/startup tests pass in 2.93 seconds, covering both formats, archived initialization with no generation-zero pin, fresh-disk restoration without another namespace, missing retained metadata refusal and deployment/workspace exclusions: **267 unique focused Rust cases**. Focused and child-process reruns are excluded. Workspace/all-target Clippy passes with warnings denied in 23.77 seconds; the server binary builds in 40.93 seconds. Formatting/diff checks, all 415 frozen Rust-source hashes, local documentation links, the protected checkout/archive and exact Cellule pins pass. No stack, SDK lifetime, deadline, resource or capacity threshold is widened. Complete initial Begin/Claim/Renew and pre-registration process loss, background service reconstruction for older orphan attempts, production producers/readers/final DDL, typed collection/backup/isolated restore and full-history/team capacity remain open. This is a local unreleasable checkpoint. + +## Exact final initialization recovery in the local cutover + +Command 31 now requires authenticated exact registration before its domain action. The existing `Record`, `Bundle`, `SavedCommand`, `Journal`, `Frame`, immutable body/header roots and independent preparation pin carry `Kind::Initialization`; there is no separate durable queue or metadata authority. The unpublished protocol purpose advances to v4, initialization codec to 2 and recovery-registration codec to 4, with no compatibility decoder. Other published final-command codecs remain unchanged. + +`PreparedCatalog::ready_initialization` freezes the private proof and original SDK command while retaining its owner. After durable registration it binds into the existing account-fair dispatcher, charging 20 KiB through uncertainty. Actual startup uses its already bounded tracked transition admission and retains the same original session through exact registered completion. Pending startup discovers the current operation's registered original before Begin; only a known original Stale/Expired denial permits a new Claim. Ready startup continues to verify the retained initialization without another admission. + +Original phase results and sequences commit atomically with catalog/ref initialization or a definitive typed denial. Unregistered/competing identities leave SDK and domain/phase state absent. Late SQL failure rolls back every effect. Cold final initialization restores only after authoritative SDK absence and delegates current owner/Admin/expiry/pin/checkpoint/pristine checks to that exact receiver. It performs no new preparation/native work and does not require a fresh Write query that could hide a definitive expiry/revocation denial. Live bound work still checks its original shared clock and lifecycle fence. Known journal knowledge wins before body/custody reads and retains the original receipt after SDK expiry or owner loss. + +The unregistered-command regression first publishes generation one against the old code, then passes with SDK/domain absence after the receiver gate. A cold expiry regression first stops at an inactive custody query, then passes with the definitive original denial after the recovery change. Eight initialization families cover both formats, late rollback, competing identity, lost registration acknowledgement, lost final acknowledgement, removal of local SQLite before real owner restoration, and original receipt recovery after SDK expiry, current permission revocation and saved-body loss. The original live/fair admission retains its 20 KiB reservation through uncertainty and closed-service recovery. + +The broader audit exposes one shared policy initializer that still submits command 31 without registration, plus queries that assume only one recovery pin exists. The initializer now registers first. Native recovery/retirement fixtures select the exact original attempt and verify unrelated recovery headers/phases remain unchanged. The final complete publication run passes **258 tests in 173.96 seconds** with four threads and standard stacks. Three actual workspace/startup tests pass in 2.57 seconds, including real registered initialization, SHA-1/SHA-256 creation, fresh-disk restoration and missing-root refusal: **261 unique focused Rust cases**. Focused reruns are excluded. Workspace/all-target Clippy passes with warnings denied in 31.97 seconds; the server builds in 33.76 seconds. Production and integration sources are unchanged after those integration/build checks; subsequent edits affect four qualification fixture files only. Formatting, diff checks, all 414 frozen Rust-source hashes, local documentation links, the protected checkout/archive and exact Cellule pin checks pass. No stack, SDK lifetime, deadline, resource or capacity threshold is widened. + +At this earlier registration checkpoint, initial Begin/Claim/Renew, pre-registration process loss and initial terminal retirement remain release blockers. Retaining the initialization pin indefinitely would prevent later catalog generation collection; the following local retirement increment removes the selected closed initial pin. The full production cutover, provider and repository/team capacity gates remain open. This checkpoint is local and unpublished. ## Mandatory publication registration qualified locally @@ -14,7 +847,7 @@ The ten standalone fixtures now use actual native packs in admitted namespaces, The final frozen-source macOS ARM64 workspace passes **655 unique Rust tests**: 531 library tests (6 Git-format / 14 object-storage / 511 server), 104 multi-server tests and 20 CLI/contract/recovery/Smart HTTP tests. The server library completes in 235.55 seconds and multi-server in 376.05 seconds. Eight isolated RustFS compatibility cases pass; the large-transfer case remains a dedicated-volume gate. All 96 Python qualification tests pass in 40.924 seconds, and workspace/all-target Clippy passes with warnings denied in 24.00 seconds. The server binary builds in 80 seconds. Formatting, diff, 409 frozen Rust-source hashes, exact dependency and protected-checkout checks pass. Earlier stack overflows are avoided through owned qualification and synchronous future construction, reducing the common ARM64 debug native poll frame from roughly 951 KiB to 707 KiB with standard stacks. A retirement expiry assertion failed once; its isolated control passed four milliseconds after expiry with unchanged identity. The fixture now verifies actual wall-clock expiry before resolution; the original timing cause remains unproven. Probes are removed, and no stacks, SDK lifetimes, production deadlines, resource limits or capacity thresholds were widened. -The increment is prepared for a new PR to main after merged #32. Exact-head Linux/provider CI is required for publication qualification. Production registration, startup/HTTP/SSH/generated producers and readers, and fresh-schema hard cutover remain the highest-priority next work. Initial/Claim/Renew/denied/pre-admission recovery, retained-input adoption/repreparation, typed collection/backup/isolated restore, resource containment, accelerated reads/physical rewriting, fair continuous maintenance and full repository/team qualification also remain open. The full objective remains open. +PR #33 is merged at `9438bb865959fb975d5349ba8b9908b461653821`. Both exact-head Linux [push Verify](https://github.com/crabbuild/canopy/actions/runs/37164049029) and [PR Verify](https://github.com/crabbuild/canopy/actions/runs/37164077931) pass at `32f5559216432c0437ac3e864d71c454ee779e1e`, each with 659 unique workspace tests, all eight RustFS compatibility cases, harness, formatting, all-target Clippy and server build. The original cancellation and all four controlled-fork cases pass. Main and the published head have the identical full tree `aacb41ee48e953cf106c75f8319667fa83639b96`, proving every published registration change is included. These results qualify that tree, not the current local format changes. Production registration, startup/HTTP/SSH/generated producers and readers, and fresh-schema hard cutover remain the highest-priority next work. Initial/Claim/Renew/denied/pre-admission recovery, retained-input adoption/repreparation, typed collection/backup/isolated restore, resource containment, accelerated reads/physical rewriting, fair continuous maintenance and full repository/team qualification also remain open. The full objective remains open. ## Linux listener ownership qualified at PR #32 diff --git a/scripts/benchmark_large_repository.py b/scripts/benchmark_large_repository.py index bab54e42..8ffe8966 100644 --- a/scripts/benchmark_large_repository.py +++ b/scripts/benchmark_large_repository.py @@ -114,7 +114,7 @@ def sample(): "process_tree_rss_bytes": rss}) + "\n") # MAX(sequence) uses the integer primary key. Avoid # table scans or long transactions on the live Cell. - for database in (args.state_dir / "node" / "runtime-v1").glob("*/repository.sqlite"): + for database in (args.state_dir / "node" / "canopy-pack-v1").glob("*/repository.sqlite"): try: connection = sqlite3.connect(database.as_uri() + "?mode=ro", uri=True, timeout=0.2) try: @@ -258,7 +258,7 @@ def api(path, payload=None, method=None): git("prepare-pull-client", "clone", "--shared", "--single-branch", "--branch", default_ref.removeprefix("refs/heads/"), str(args.work_dir / "warm-v2.git"), str(pull_work)) git("configure-pull-client", "-C", str(pull_work), "remote", "set-url", "origin", url) - # The binary's startup contract removes runtime-v1 before recovering + # The binary's startup contract removes canopy-pack-v1 before recovering # authoritative Cells from the provider; no manual deletion is needed. commit_env = {**git_env, "GIT_AUTHOR_NAME": "Canopy Evaluation", "GIT_AUTHOR_EMAIL": "evaluation@example.invalid", diff --git a/scripts/smoke_s3_cache.py b/scripts/smoke_s3_cache.py index 05f5add2..f69bbb17 100644 --- a/scripts/smoke_s3_cache.py +++ b/scripts/smoke_s3_cache.py @@ -113,7 +113,7 @@ def clone(name, protocol): def cached_objects(data): objects = {} - for cache in (data / "runtime-v1").glob("canopy-git-*/repo.git"): + for cache in (data / "canopy-pack-v1").glob("canopy-git-*/repo.git"): if (cache / "objects/info/alternates").exists(): continue for path in (cache / "objects").glob("*/*"): diff --git a/scripts/smoke_s3_process.py b/scripts/smoke_s3_process.py index c388fb07..4eb932a9 100644 --- a/scripts/smoke_s3_process.py +++ b/scripts/smoke_s3_process.py @@ -755,7 +755,7 @@ def main(): clone_and_verify(f"{base_url}/canopy/other.git", directory / "clean-other", other_oid, other_readme) second.kill() second.wait(timeout=10) - abandoned = directory / "second" / "runtime-v1" + abandoned = directory / "second" / "canopy-pack-v1" stale_caches = list(abandoned.glob("canopy-git-*")) assert stale_caches, "expected a retained fetch cache at owner death" sentinel = abandoned / "abandoned-upload"