diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 717a84f6d..d76f27f62 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -298,14 +298,18 @@ jobs: # The control plane for vector indexes is not reachable over the wire while # this backend declares no vector search capability, so its tests drive the # storage layer directly against this job's PostgreSQL. The PutItem create - # race tests hold a create open in an outside transaction, and the - # write-transaction conflict tests force a deadlock; both also need direct - # access. They build their own throwaway databases; the connection string is - # the server, not a database. + # race tests hold a create open in an outside transaction, the + # write-transaction conflict tests force a deadlock, and the backup restore + # cases need catalog rows the current binary would not write; all need + # direct access. They build their own throwaway databases; the connection + # string is the server, not a database. - name: Run PostgreSQL storage-level tests env: EXTENDDB_TEST_PG_CONNECTION_STRING: postgresql://postgres:devpass@127.0.0.1:5432 - run: cargo test --release -p extenddb-storage-postgres --test vector_control_plane --test put_create_race --test twi_conflict + run: >- + cargo test --release -p extenddb-storage-postgres + --test vector_control_plane --test put_create_race --test twi_conflict + --test backup_restore # The daemonized server logs to syslog; dump it so server-side failures # are diagnosable from the job log. diff --git a/crates/app/src/cmd_migrate.rs b/crates/app/src/cmd_migrate.rs index da2e3c376..ebf085db1 100755 --- a/crates/app/src/cmd_migrate.rs +++ b/crates/app/src/cmd_migrate.rs @@ -90,6 +90,7 @@ async fn apply_migrations( println!(" Current version: {current_display}"); let expected = bootstrap.expected_catalog_version(); + refuse_newer_catalog(current.as_deref(), &expected)?; let catalog_pending = current.as_deref() != Some(expected.as_str()); // Data migrations are tracked in the data database's own ledger, separate @@ -210,3 +211,46 @@ async fn apply_migrations( Ok(()) } + +/// A catalog written by a newer binary is not a pending upgrade. The runners +/// write the compiled version after every walk, so letting an older binary +/// "migrate" it would stamp the older version onto the newer schema and let +/// that binary start against tables it does not understand. +fn refuse_newer_catalog(current: Option<&str>, expected: &str) -> anyhow::Result<()> { + use extenddb_core::version::CatalogVersion; + let (Some(current), Ok(expected_v)) = (current, expected.parse::()) else { + return Ok(()); + }; + // A malformed stored version is reported by the migration itself. + let Ok(current_v) = current.parse::() else { + return Ok(()); + }; + if current_v > expected_v { + anyhow::bail!( + "catalog version {current} is newer than this binary's {expected}. \ + extenddb migrate does not downgrade a catalog; run a binary that \ + expects {current} or later." + ); + } + Ok(()) +} + +#[cfg(test)] +mod newer_catalog_tests { + use super::refuse_newer_catalog; + + #[test] + fn older_or_equal_or_absent_is_allowed() { + assert!(refuse_newer_catalog(None, "0.0.4").is_ok()); + assert!(refuse_newer_catalog(Some("0.0.3"), "0.0.4").is_ok()); + assert!(refuse_newer_catalog(Some("0.0.4"), "0.0.4").is_ok()); + assert!(refuse_newer_catalog(Some("garbage"), "0.0.4").is_ok()); + } + + #[test] + fn newer_is_refused() { + let err = refuse_newer_catalog(Some("0.0.5"), "0.0.4").expect_err("newer catalog"); + assert!(err.to_string().contains("does not downgrade"), "{err}"); + assert!(refuse_newer_catalog(Some("0.1.0"), "0.0.4").is_err()); + } +} diff --git a/crates/core/src/error/mod.rs b/crates/core/src/error/mod.rs index c0a28c0b3..e950fe84d 100755 --- a/crates/core/src/error/mod.rs +++ b/crates/core/src/error/mod.rs @@ -24,6 +24,8 @@ pub enum DynamoDbError { #[error("{0}")] BackupNotFoundException(String), #[error("{0}")] + BackupInUseException(String), + #[error("{0}")] ResourceInUseException(String), /// A per-table or per-account limit was exceeded. /// @@ -114,6 +116,7 @@ impl DynamoDbError { Self::ValidationException(_) | Self::ResourceNotFoundException(_) | Self::BackupNotFoundException(_) + | Self::BackupInUseException(_) | Self::ResourceInUseException(_) | Self::LimitExceededException(_) | Self::ConditionalCheckFailedException(..) @@ -155,6 +158,7 @@ impl DynamoDbError { Self::ValidationException(_) => "ValidationException", Self::ResourceNotFoundException(_) => "ResourceNotFoundException", Self::BackupNotFoundException(_) => "BackupNotFoundException", + Self::BackupInUseException(_) => "BackupInUseException", Self::ResourceInUseException(_) => "ResourceInUseException", Self::LimitExceededException(_) => "LimitExceededException", Self::ConditionalCheckFailedException(..) => "ConditionalCheckFailedException", @@ -221,6 +225,7 @@ impl DynamoDbError { Self::ValidationException(m) | Self::ResourceNotFoundException(m) | Self::BackupNotFoundException(m) + | Self::BackupInUseException(m) | Self::ResourceInUseException(m) | Self::LimitExceededException(m) | Self::ConditionalCheckFailedException(m, _) @@ -317,6 +322,7 @@ mod tests { (DynamoDbError::ResourceInUseException(String::new()), 400), (DynamoDbError::ResourceNotFoundException(String::new()), 400), (DynamoDbError::BackupNotFoundException(String::new()), 400), + (DynamoDbError::BackupInUseException(String::new()), 400), (DynamoDbError::SerializationException(String::new()), 400), (DynamoDbError::ServiceUnavailable(String::new()), 503), (DynamoDbError::ThrottlingException(String::new()), 400), diff --git a/crates/core/src/validation/mod.rs b/crates/core/src/validation/mod.rs index 5b780fd6a..ee0b25972 100755 --- a/crates/core/src/validation/mod.rs +++ b/crates/core/src/validation/mod.rs @@ -886,7 +886,14 @@ fn format_attr_defs(defs: &[AttributeDefinition]) -> String { .join(", ") } -fn validate_provisioned_throughput(input: &CreateTableInput) -> Result<(), DynamoDbError> { +/// Validate table-level throughput against the effective billing mode. +/// +/// # Errors +/// +/// Returns [`DynamoDbError::ValidationException`] when required provisioned +/// throughput is absent or invalid, or when throughput is supplied for an +/// on-demand table. +pub fn validate_provisioned_throughput(input: &CreateTableInput) -> Result<(), DynamoDbError> { let billing = input.billing_mode.unwrap_or(BillingMode::Provisioned); match billing { BillingMode::Provisioned => { diff --git a/crates/core/src/version.rs b/crates/core/src/version.rs index 38edf0f63..6387f91e7 100755 --- a/crates/core/src/version.rs +++ b/crates/core/src/version.rs @@ -27,7 +27,7 @@ use std::str::FromStr; /// let parsed: CatalogVersion = "1.2.0".parse().unwrap(); /// assert_eq!(parsed, v); /// ``` -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] pub struct CatalogVersion { major: u32, minor: u32, @@ -125,6 +125,15 @@ impl FromStr for CatalogVersion { #[cfg(test)] mod tests { + #[test] + fn orders_by_component() { + use super::CatalogVersion; + assert!(CatalogVersion::new(0, 0, 3) < CatalogVersion::new(0, 0, 4)); + assert!(CatalogVersion::new(0, 0, 10) > CatalogVersion::new(0, 0, 9)); + assert!(CatalogVersion::new(0, 1, 0) > CatalogVersion::new(0, 0, 99)); + assert!(CatalogVersion::new(1, 0, 0) > CatalogVersion::new(0, 99, 99)); + } + use super::*; #[test] diff --git a/crates/engine/src/backup.rs b/crates/engine/src/backup.rs index 510190d1e..e00af4eb2 100755 --- a/crates/engine/src/backup.rs +++ b/crates/engine/src/backup.rs @@ -130,11 +130,127 @@ pub(crate) async fn handle_delete_backup( serialize_output(&json!({ "BackupDescription": desc })) } +const UNSUPPORTED_RESTORE_TABLE_OVERRIDE_FIELDS: [&str; 5] = [ + "GlobalSecondaryIndexOverride", + "LocalSecondaryIndexOverride", + "SSESpecificationOverride", + "OnDemandThroughputOverride", + "VectorIndexOverride", +]; + +#[derive(Debug, Clone, PartialEq)] +struct RestoreTableOverrides { + billing_mode: Option, + provisioned_throughput: Option, +} + +impl RestoreTableOverrides { + fn is_empty(&self) -> bool { + self.billing_mode.is_none() && self.provisioned_throughput.is_none() + } + + fn effective_billing_mode( + &self, + source_billing_mode: Option<&str>, + ) -> extenddb_core::types::BillingMode { + self.billing_mode.unwrap_or_else(|| { + if source_billing_mode == Some("PAY_PER_REQUEST") { + extenddb_core::types::BillingMode::PayPerRequest + } else { + extenddb_core::types::BillingMode::Provisioned + } + }) + } + + /// The rules that depend only on the request itself: a PROVISIONED + /// override needs a throughput, and a throughput must be well formed. + /// Checked before anything is read from storage. + fn validate_shape(&self) -> Result<(), DynamoDbError> { + use extenddb_core::types::BillingMode; + + if matches!(self.billing_mode, Some(BillingMode::Provisioned)) + && self.provisioned_throughput.is_none() + { + return Err(DynamoDbError::ValidationException( + "One or more parameter values were invalid: ProvisionedThroughputOverride must \ + be specified when BillingModeOverride is PROVISIONED" + .to_owned(), + )); + } + + if let Some(throughput) = &self.provisioned_throughput { + let input = extenddb_core::types::CreateTableInput { + billing_mode: Some(BillingMode::Provisioned), + provisioned_throughput: Some(throughput.clone()), + ..Default::default() + }; + extenddb_core::validation::validate_provisioned_throughput(&input)?; + } + Ok(()) + } + + /// The rule that needs the backup's own billing mode: a throughput + /// override is only meaningful if the restored table is PROVISIONED. + fn validate_for_source(&self, source_billing_mode: Option<&str>) -> Result<(), DynamoDbError> { + use extenddb_core::types::BillingMode; + + if self.provisioned_throughput.is_some() + && self.effective_billing_mode(source_billing_mode) == BillingMode::PayPerRequest + { + return Err(DynamoDbError::ValidationException( + "One or more parameter values were invalid: ProvisionedThroughputOverride can \ + only be specified with BillingModeOverride PROVISIONED" + .to_owned(), + )); + } + Ok(()) + } +} + +fn parse_restore_table_overrides(body: &Value) -> Result { + crate::validate_enum_fields( + body, + &[crate::EnumField { + json_name: "BillingModeOverride", + valid: &["PROVISIONED", "PAY_PER_REQUEST"], + clause: crate::EnumClause::Named("billingModeOverride"), + }], + )?; + + for field in UNSUPPORTED_RESTORE_TABLE_OVERRIDE_FIELDS { + if body.get(field).is_some() { + return Err(DynamoDbError::ValidationException(format!( + "RestoreTableFromBackup does not support {field} on ExtendDB" + ))); + } + } + + let billing_mode = body + .get("BillingModeOverride") + .cloned() + .map(serde_json::from_value) + .transpose() + .map_err(crate::deserialize_error)?; + let provisioned_throughput = body + .get("ProvisionedThroughputOverride") + .cloned() + .map(serde_json::from_value) + .transpose() + .map_err(crate::deserialize_error)?; + + Ok(RestoreTableOverrides { + billing_mode, + provisioned_throughput, + }) +} + /// Handle `RestoreTableFromBackup`. pub(crate) async fn handle_restore_table_from_backup( body: Value, ctx: &OperationContext, ) -> Result { + let overrides = parse_restore_table_overrides(&body)?; + let target_table_name = body .get("TargetTableName") .and_then(|v| v.as_str()) @@ -145,27 +261,71 @@ pub(crate) async fn handle_restore_table_from_backup( .to_owned(), ) })?; + extenddb_core::validation::validate_table_name(target_table_name, &ctx.limits)?; let backup_arn = backup_arn_field(&body, &ctx.account_id)?; + // Rules that need nothing from the backup come first, so a malformed + // request is refused before any storage read and before a missing backup + // can take precedence over the request's own error. + overrides.validate_shape()?; + let source_billing_mode = if overrides.is_empty() { + None + } else { + ctx.storage + .describe_backup(&ctx.account_id, &backup_arn) + .await + .map_err(storage_err_to_dynamo)? + .source_table_details + .billing_mode + }; + overrides.validate_for_source(source_billing_mode.as_deref())?; + let effective_billing_mode = overrides.effective_billing_mode(source_billing_mode.as_deref()); + + let storage_overrides = extenddb_storage::RestoreTableOverrides { + billing_mode: overrides.billing_mode, + provisioned_throughput: overrides.provisioned_throughput.clone(), + }; let mut desc = ctx .storage - .restore_table_from_backup(&ctx.account_id, target_table_name, &backup_arn) + .restore_table_from_backup( + &ctx.account_id, + target_table_name, + &backup_arn, + storage_overrides, + ) .await .map_err(storage_err_to_dynamo)?; + if effective_billing_mode == extenddb_core::types::BillingMode::PayPerRequest + && !overrides.is_empty() + { + // Backends may retain catalog capacity fields while creating an + // on-demand restore target. The response must expose normalized zero + // values, and subsequent backups normalize retained values before storage. + desc.provisioned_throughput = Default::default(); + for index in desc.global_secondary_indexes.iter_mut().flatten() { + index.provisioned_throughput = Some(Default::default()); + } + } + // The restore response's TableDescription reports where the data came // from and that the restore is under way: SourceBackupArn and // RestoreInProgress: true, pinned by the ground-truth runs of 2026-08-24 - // (us-east-1 and eu-west-2). Set here rather than in each backend because - // the summary is response metadata about this call, not table state the - // backends persist. + // (us-east-1 and eu-west-2). The service returns the table CREATING with + // the restore in progress; the backends report CREATING here too, whatever + // the copy has reached. The time is the backend's own record where it + // keeps one, so the response and later DescribeTable calls agree. let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_secs_f64()) .unwrap_or_default(); + let restore_date_time = desc + .restore_summary + .as_ref() + .map_or(now, |r| r.restore_date_time); desc.restore_summary = Some(extenddb_core::types::RestoreSummary { source_backup_arn: Some(backup_arn.clone()), - restore_date_time: now, + restore_date_time, restore_in_progress: true, }); @@ -274,6 +434,13 @@ fn storage_err_to_dynamo(e: extenddb_storage::error::StorageError) -> DynamoDbEr extenddb_storage::error::StorageError::TableAlreadyExists(msg) => { DynamoDbError::ResourceInUseException(msg) } + extenddb_storage::error::StorageError::TableNotActive(msg) + | extenddb_storage::error::StorageError::IndexesInUse(msg) => { + DynamoDbError::ResourceInUseException(msg) + } + extenddb_storage::error::StorageError::NoOpUpdate(msg) => { + DynamoDbError::ValidationException(msg) + } extenddb_storage::error::StorageError::Validation(msg) => { // A missing (or deleted) backup surfaces from the backend as a // Validation error carrying "Backup not found"; DynamoDB reports @@ -292,6 +459,12 @@ fn storage_err_to_dynamo(e: extenddb_storage::error::StorageError) -> DynamoDbEr extenddb_storage::error::StorageError::Unsupported(msg) => { DynamoDbError::ValidationException(msg) } + extenddb_storage::error::StorageError::LimitExceeded(msg) => { + DynamoDbError::LimitExceededException(msg) + } + extenddb_storage::error::StorageError::BackupInUse(msg) => { + DynamoDbError::BackupInUseException(msg) + } other => { tracing::error!(internal_error = %other, "backup storage error"); DynamoDbError::InternalServerError("Internal server error".to_owned()) @@ -301,10 +474,11 @@ fn storage_err_to_dynamo(e: extenddb_storage::error::StorageError) -> DynamoDbEr #[cfg(test)] mod tests { - use super::{backup_arn_field, storage_err_to_dynamo}; + use super::{backup_arn_field, parse_restore_table_overrides, storage_err_to_dynamo}; use extenddb_core::error::DynamoDbError; + use extenddb_core::types::{BillingMode, ProvisionedThroughput}; use extenddb_storage::error::StorageError; - use serde_json::json; + use serde_json::{Value, json}; const ACCOUNT: &str = "123456789012"; @@ -384,6 +558,161 @@ mod tests { ); } + #[test] + fn restore_request_without_overrides_is_accepted() { + let body = json!({ + "TargetTableName": "MusicRestored", + "BackupArn": arn(ACCOUNT), + }); + assert!(parse_restore_table_overrides(&body).unwrap().is_empty()); + } + + fn throughput(read: i64, write: i64) -> ProvisionedThroughput { + ProvisionedThroughput { + read_capacity_units: read, + write_capacity_units: write, + } + } + + #[test] + fn provisioned_override_with_throughput_is_accepted() { + let overrides = parse_restore_table_overrides(&json!({ + "BillingModeOverride": "PROVISIONED", + "ProvisionedThroughputOverride": { + "ReadCapacityUnits": 5, + "WriteCapacityUnits": 5, + }, + })) + .unwrap(); + overrides + .validate_for_source(Some("PAY_PER_REQUEST")) + .unwrap(); + + assert_eq!(overrides.billing_mode, Some(BillingMode::Provisioned)); + assert_eq!(overrides.provisioned_throughput, Some(throughput(5, 5))); + } + + #[test] + fn provisioned_override_without_throughput_is_refused() { + let overrides = parse_restore_table_overrides(&json!({ + "BillingModeOverride": "PROVISIONED", + })) + .unwrap(); + let err = overrides.validate_shape().unwrap_err(); + assert!(matches!(err, DynamoDbError::ValidationException(_))); + } + + #[test] + fn throughput_override_under_pay_per_request_is_refused() { + let overrides = parse_restore_table_overrides(&json!({ + "BillingModeOverride": "PAY_PER_REQUEST", + "ProvisionedThroughputOverride": { + "ReadCapacityUnits": 5, + "WriteCapacityUnits": 5, + }, + })) + .unwrap(); + let err = overrides + .validate_for_source(Some("PROVISIONED")) + .unwrap_err(); + match err { + DynamoDbError::ValidationException(message) => assert!(message.contains( + "ProvisionedThroughputOverride can only be specified with \ + BillingModeOverride PROVISIONED" + )), + other => panic!("expected ValidationException, got {other:?}"), + } + } + + #[test] + fn invalid_provisioned_throughput_override_matches_create_table() { + let expected = "One or more parameter values were invalid: ReadCapacityUnits and \ + WriteCapacityUnits must both be greater than or equal to 1 for table"; + for (read, write) in [(0, 5), (5, 0), (-1, 5), (5, -1)] { + let overrides = parse_restore_table_overrides(&json!({ + "BillingModeOverride": "PROVISIONED", + "ProvisionedThroughputOverride": { + "ReadCapacityUnits": read, + "WriteCapacityUnits": write, + }, + })) + .unwrap(); + let err = overrides.validate_shape().unwrap_err(); + match err { + DynamoDbError::ValidationException(message) => assert_eq!(message, expected), + other => panic!("expected ValidationException, got {other:?}"), + } + } + } + + #[test] + fn missing_provisioned_throughput_override_member_matches_create_table() { + for (member, body) in [ + ( + "ReadCapacityUnits", + json!({ + "BillingModeOverride": "PROVISIONED", + "ProvisionedThroughputOverride": { "WriteCapacityUnits": 5 }, + }), + ), + ( + "WriteCapacityUnits", + json!({ + "BillingModeOverride": "PROVISIONED", + "ProvisionedThroughputOverride": { "ReadCapacityUnits": 5 }, + }), + ), + ] { + let err = parse_restore_table_overrides(&body).unwrap_err(); + match err { + DynamoDbError::SerializationException(message) => { + assert!( + message.contains(&format!("missing field `{member}`")), + "{message}" + ); + } + other => panic!("expected SerializationException, got {other:?}"), + } + } + } + + fn assert_restore_override_rejected(field: &str) { + let mut body = json!({ + "TargetTableName": "MusicRestored", + "BackupArn": arn(ACCOUNT), + }); + body[field] = Value::Null; + let err = parse_restore_table_overrides(&body).unwrap_err(); + match err { + DynamoDbError::ValidationException(message) => assert_eq!( + message, + format!("RestoreTableFromBackup does not support {field} on ExtendDB") + ), + other => panic!("expected ValidationException for {field}, got {other:?}"), + } + } + + #[test] + fn index_and_sse_overrides_remain_refused() { + for field in [ + "GlobalSecondaryIndexOverride", + "LocalSecondaryIndexOverride", + "SSESpecificationOverride", + ] { + assert_restore_override_rejected(field); + } + } + + #[test] + fn on_demand_throughput_override_is_refused() { + assert_restore_override_rejected("OnDemandThroughputOverride"); + } + + #[test] + fn vector_index_override_is_refused() { + assert_restore_override_rejected("VectorIndexOverride"); + } + #[test] fn account_prefix_does_not_match() { // A shorter account id that is a prefix of the caller's must not pass. diff --git a/crates/engine/src/create_table.rs b/crates/engine/src/create_table.rs index 257cd2299..019147f3d 100755 --- a/crates/engine/src/create_table.rs +++ b/crates/engine/src/create_table.rs @@ -134,6 +134,7 @@ pub(crate) fn storage_err_to_dynamo(e: extenddb_storage::error::StorageError) -> // state: the documented delete-table sentence and the measured // phase-dependent vector refusal both arrive through this one arm. StorageError::IndexesInUse(msg) => DynamoDbError::ResourceInUseException(msg), + StorageError::BackupInUse(msg) => DynamoDbError::BackupInUseException(msg), StorageError::LimitExceeded(msg) => DynamoDbError::LimitExceededException(msg), // Retryable by definition, so it maps like Connection: a 503 the SDKs // retry, rather than a 500 they surface. StorageError::Transient(msg) => { diff --git a/crates/storage-mongodb/src/backup_engine.rs b/crates/storage-mongodb/src/backup_engine.rs index 5476c7cc0..29808bc63 100644 --- a/crates/storage-mongodb/src/backup_engine.rs +++ b/crates/storage-mongodb/src/backup_engine.rs @@ -26,8 +26,8 @@ use extenddb_core::types::{ PointInTimeRecoveryDescription, Projection, ProvisionedThroughput, SourceTableDetails, TableDescription, }; -use extenddb_storage::BackupEngine; use extenddb_storage::error::StorageError; +use extenddb_storage::{BackupEngine, RestoreTableOverrides}; use crate::MongoEngine; use crate::data::data_collection_name; @@ -102,6 +102,251 @@ fn restore_provisioned_throughput( } } +/// GSI throughput follows the table's billing mode the same way: a table +/// switched from PROVISIONED to PAY_PER_REQUEST keeps its indexes' old +/// capacities in the catalog, and restoring them would make the restored +/// on-demand table's indexes report capacity a fresh one does not. +fn restore_gsi_throughput( + billing_mode: &str, + gsis: Option>, +) -> Option> { + if billing_mode == "PROVISIONED" { + return gsis; + } + gsis.map(|gsis| { + gsis.into_iter() + .map(|mut gsi| { + gsi.provisioned_throughput = None; + + gsi + }) + .collect() + }) +} + +/// A restore first claims backup metadata, then transfers protection to the +/// provenance-bearing CREATING target. DeleteBackup claims the same metadata +/// before dropping data, so neither operation can pass the other between its +/// check and destructive action. +fn restore_claim_filter(account_id: &str, backup_arn: &str) -> Document { + doc! { + "_id": backup_arn, + "account_id": account_id, + "backup_status": "AVAILABLE", + "restore_in_progress": { "$exists": false }, + "delete_in_progress": { "$exists": false }, + } +} + +fn delete_claim_filter(account_id: &str, backup_arn: &str) -> Document { + doc! { + "_id": backup_arn, + "account_id": account_id, + "backup_status": "AVAILABLE", + "restore_in_progress": { "$exists": false }, + "delete_in_progress": { "$exists": false }, + } +} + +fn restoring_target_filter(account_id: &str, backup_arn: &str) -> Document { + doc! { + "_id.account_id": account_id, + "restore_source_backup_arn": backup_arn, + "table_status": "CREATING", + } +} + +struct BackupMetadataClaim { + backups: mongodb::Collection, + account_id: String, + backup_arn: String, + field: &'static str, + operation_id: String, + armed: bool, +} + +impl BackupMetadataClaim { + fn new( + backups: mongodb::Collection, + account_id: &str, + backup_arn: &str, + field: &'static str, + operation_id: String, + ) -> Self { + Self { + backups, + account_id: account_id.to_owned(), + backup_arn: backup_arn.to_owned(), + field, + operation_id, + armed: true, + } + } + + fn filter(&self) -> Document { + let mut filter = doc! { + "_id": &self.backup_arn, + "account_id": &self.account_id, + }; + filter.insert(self.field, &self.operation_id); + filter + } + + fn unset_update(field: &'static str) -> Document { + let mut unset = Document::new(); + unset.insert(field, ""); + doc! { "$unset": unset } + } + + async fn release(&mut self) -> Result<(), StorageError> { + if !self.armed { + return Ok(()); + } + self.backups + .update_one(self.filter(), Self::unset_update(self.field)) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + self.armed = false; + Ok(()) + } + + fn disarm(&mut self) { + self.armed = false; + } +} + +impl Drop for BackupMetadataClaim { + fn drop(&mut self) { + if !self.armed { + return; + } + let backups = self.backups.clone(); + let filter = self.filter(); + let update = Self::unset_update(self.field); + let backup_arn = self.backup_arn.clone(); + if let Ok(runtime) = tokio::runtime::Handle::try_current() { + let _cleanup = runtime.spawn(async move { + if let Err(error) = backups.update_one(filter, update).await { + tracing::error!( + "could not clear cancelled backup metadata claim for {backup_arn}: {error}" + ); + } + }); + } + } +} + +impl MongoEngine { + /// Clear the backup-metadata claims a process that died mid-operation + /// left behind. Both markers belong to one process: `restore_in_progress` + /// is held only until the restore's target exists (after that the + /// CREATING target itself keeps DeleteBackup out), and `delete_in_progress` + /// only until the drop and the DELETED stamp. Neither survives a crash + /// usefully, and left in place each would make the backup undeletable or + /// unrestorable for good. Run at startup, before requests are served, so + /// no live claim can be swept. + /// + /// A stale restore claim is simply removed: the restore never created its + /// target (the marker is released once it has), so there is nothing else + /// to undo. A stale delete claim is finished rather than released: the + /// caller already received an answer or died waiting for one, the + /// collection drop is idempotent, and leaving the backup AVAILABLE could + /// point at data the crash already dropped. + pub async fn sweep_stale_backup_claims(&self) -> Result<(usize, usize), StorageError> { + let backups_coll = self.catalog_db.collection::("backups"); + let restores = backups_coll + .update_many( + doc! { "restore_in_progress": { "$exists": true } }, + doc! { "$unset": { "restore_in_progress": "" } }, + ) + .await + .map_err(|e| StorageError::Internal(e.to_string()))? + .modified_count; + + let mut deletes = 0usize; + let mut pending = backups_coll + .find(doc! { "delete_in_progress": { "$exists": true } }) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + while let Some(meta) = pending + .try_next() + .await + .map_err(|e| StorageError::Internal(e.to_string()))? + { + if let Ok(backup_id) = meta.get_str("backup_id") { + self.drop_collection_if_exists(&backup_collection_name(backup_id)) + .await?; + } + let id = meta + .get_str("_id") + .map_err(|_| StorageError::Internal("backup without _id".to_owned()))?; + backups_coll + .update_one( + doc! { "_id": id }, + doc! { + "$set": { "backup_status": "DELETED" }, + "$unset": { "delete_in_progress": "" }, + }, + ) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + deletes += 1; + } + Ok((usize::try_from(restores).unwrap_or(usize::MAX), deletes)) + } +} + +impl MongoEngine { + /// Remove a partial restore target without allowing its provenance to keep + /// the source backup in use. The full predicate ensures cleanup can only + /// claim the target created by this restore. + async fn abort_backup_restore( + &self, + account_id: &str, + target_table_name: &str, + backup_arn: &str, + restore_operation_id: &str, + ) -> Result { + let removed = self + .catalog_db + .collection::("tables") + .find_one_and_delete(doc! { + "_id": { "account_id": account_id, "table_name": target_table_name }, + "restore_source_backup_arn": backup_arn, + "restore_operation_id": restore_operation_id, + "table_status": "CREATING", + "status_transition_at": bson::Bson::Null, + }) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + let Some(removed) = removed else { + return Ok(false); + }; + let table_id = removed + .get_str("table_id") + .map_err(|_| StorageError::Internal("restore target missing table_id".to_owned()))?; + + // The catalog row is already gone, so even a failed physical cleanup + // cannot leave provenance that blocks DeleteBackup. Try every cleanup + // step before returning the first error so retries have less to do. + let mut cleanup_error = None; + let collection = data_collection_name(table_id); + if let Err(e) = self.drop_collection_if_exists(&collection).await { + cleanup_error = Some(e); + } + if let Err(e) = self.drop_index_collections_for_table(table_id).await { + cleanup_error.get_or_insert(e); + } + self.gsi_cache_invalidate(table_id); + + if let Some(e) = cleanup_error { + Err(e) + } else { + Ok(true) + } + } +} + impl BackupEngine for MongoEngine { fn create_backup( &self, @@ -491,36 +736,117 @@ impl BackupEngine for MongoEngine { let backup_arn = backup_arn.to_string(); Box::pin(async move { let desc = self.describe_backup(&account_id, &backup_arn).await?; - - // Look up the physical collection name from metadata (account-scoped). let backups_coll = self.catalog_db.collection::("backups"); + let tables_coll = self.catalog_db.collection::("tables"); + + // Fast-path the durable half of the two-phase restore claim. + if let Some(target) = tables_coll + .find_one(restoring_target_filter(&account_id, &backup_arn)) + .await + .map_err(|e| StorageError::Internal(e.to_string()))? + { + let name = target + .get_document("_id") + .ok() + .and_then(|id| id.get_str("table_name").ok()) + .unwrap_or("?"); + return Err(StorageError::BackupInUse(format!( + "Backup is being used to restore table {name}: {backup_arn}" + ))); + } + + // Atomically claim deletion only when restore has not claimed the + // metadata. Restore uses the inverse predicate, so once either + // marker is installed the other operation cannot start. + let delete_operation_id = uuid::Uuid::new_v4().to_string(); let meta = backups_coll - .find_one(doc! { "_id": &backup_arn, "account_id": &account_id }) + .find_one_and_update( + delete_claim_filter(&account_id, &backup_arn), + doc! { "$set": { "delete_in_progress": &delete_operation_id } }, + ) + .return_document(mongodb::options::ReturnDocument::Before) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + let Some(meta) = meta else { + let restoring = tables_coll + .find_one(restoring_target_filter(&account_id, &backup_arn)) + .await + .map_err(|e| StorageError::Internal(e.to_string()))? + .is_some(); + let claimed_by_restore = backups_coll + .find_one(doc! { + "_id": &backup_arn, + "account_id": &account_id, + "backup_status": "AVAILABLE", + "restore_in_progress": { "$exists": true }, + }) + .await + .map_err(|e| StorageError::Internal(e.to_string()))? + .is_some(); + if restoring || claimed_by_restore { + return Err(StorageError::BackupInUse(format!( + "Backup is being used by a restore: {backup_arn}" + ))); + } + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + }; + let mut delete_claim = BackupMetadataClaim::new( + backups_coll.clone(), + &account_id, + &backup_arn, + "delete_in_progress", + delete_operation_id, + ); + + // A restore can transfer its marker to a target between the first + // target read and our metadata claim. Recheck after claiming; new + // restores are now excluded by delete_in_progress. + if let Some(target) = tables_coll + .find_one(restoring_target_filter(&account_id, &backup_arn)) .await .map_err(|e| StorageError::Internal(e.to_string()))? - .ok_or_else(|| { - StorageError::Validation(format!("Backup not found: {backup_arn}")) - })?; + { + let name = target + .get_document("_id") + .ok() + .and_then(|id| id.get_str("table_name").ok()) + .unwrap_or("?") + .to_owned(); + delete_claim.release().await?; + return Err(StorageError::BackupInUse(format!( + "Backup is being used to restore table {name}: {backup_arn}" + ))); + } - // Drop the backup collection. If backup_id is absent (e.g., a - // pre-`$out` backup on an old catalog) we skip — nothing to drop - // at the collection level in that case. + // Drop the backup collection. If backup_id is absent (for an old + // pre-$out backup), there is no physical collection to drop. if let Ok(backup_id) = meta.get_str("backup_id") { let coll_name = backup_collection_name(backup_id); let coll = self.data_db.collection::(&coll_name); - coll.drop() - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + if let Err(error) = coll.drop().await { + if let Err(cleanup) = delete_claim.release().await { + tracing::error!( + "DeleteBackup {backup_arn} failed ({error}) and its metadata claim \ + could not be cleared: {cleanup}" + ); + } + return Err(StorageError::Internal(error.to_string())); + } } - // Mark backup as deleted (account-scoped) backups_coll .update_one( - doc! { "_id": &backup_arn, "account_id": &account_id }, - doc! { "$set": { "backup_status": "DELETED" } }, + delete_claim.filter(), + doc! { + "$set": { "backup_status": "DELETED" }, + "$unset": { "delete_in_progress": "" }, + }, ) .await .map_err(|e| StorageError::Internal(e.to_string()))?; + delete_claim.disarm(); Ok(BackupDescription { backup_details: BackupDetails { @@ -537,164 +863,276 @@ impl BackupEngine for MongoEngine { account_id: &str, target_table_name: &str, backup_arn: &str, + overrides: RestoreTableOverrides, ) -> BoxFuture<'_, Result> { let account_id = account_id.to_string(); let target_table_name = target_table_name.to_string(); let backup_arn = backup_arn.to_string(); Box::pin(async move { let backups_coll = self.catalog_db.collection::("backups"); + let restore_operation_id = uuid::Uuid::new_v4().to_string(); let backup_doc = backups_coll - // Scope to the calling account (defence-in-depth: the engine - // layer already enforces ARN ownership, and describe/delete are - // account-scoped at the storage layer too). - .find_one(doc! { - "_id": &backup_arn, - "account_id": &account_id, - "backup_status": "AVAILABLE", - }) + // This metadata marker is phase one of the restore claim. It is + // installed atomically with the AVAILABLE check, before any + // restore state is derived, and excludes DeleteBackup's claim. + .find_one_and_update( + restore_claim_filter(&account_id, &backup_arn), + doc! { "$set": { "restore_in_progress": &restore_operation_id } }, + ) + .return_document(mongodb::options::ReturnDocument::After) .await .map_err(|e| StorageError::Internal(e.to_string()))? .ok_or_else(|| { StorageError::Validation(format!("Backup not found: {backup_arn}")) })?; + let mut restore_claim = BackupMetadataClaim::new( + backups_coll.clone(), + &account_id, + &backup_arn, + "restore_in_progress", + restore_operation_id.clone(), + ); - let key_schema_bson = backup_doc - .get_array("key_schema") - .map_err(|_| StorageError::Internal("missing key_schema".to_string()))?; - let attr_defs_bson = backup_doc - .get_array("attribute_definitions") - .map_err(|_| StorageError::Internal("missing attribute_definitions".to_string()))?; - let billing = backup_doc - .get_str("billing_mode") - .unwrap_or("PAY_PER_REQUEST"); + let prepared = async { + let key_schema_bson = backup_doc + .get_array("key_schema") + .map_err(|_| StorageError::Internal("missing key_schema".to_string()))?; + let attr_defs_bson = + backup_doc.get_array("attribute_definitions").map_err(|_| { + StorageError::Internal("missing attribute_definitions".to_string()) + })?; + let billing = backup_doc + .get_str("billing_mode") + .unwrap_or("PAY_PER_REQUEST"); + let effective_billing = match overrides.billing_mode { + Some(extenddb_core::types::BillingMode::Provisioned) => "PROVISIONED", + Some(extenddb_core::types::BillingMode::PayPerRequest) => "PAY_PER_REQUEST", + None => billing, + }; + + let ks_json = serde_json::to_value(key_schema_bson) + .map_err(|e| StorageError::Internal(format!("serialize key_schema: {e}")))?; + let ad_json = serde_json::to_value(attr_defs_bson) + .map_err(|e| StorageError::Internal(format!("serialize attr_defs: {e}")))?; + + let key_schema: Vec = + serde_json::from_value(ks_json) + .map_err(|e| StorageError::Internal(format!("parse key_schema: {e}")))?; + let attr_defs: Vec = + serde_json::from_value(ad_json) + .map_err(|e| StorageError::Internal(format!("parse attr_defs: {e}")))?; + + let billing_mode = if effective_billing == "PAY_PER_REQUEST" { + Some(extenddb_core::types::BillingMode::PayPerRequest) + } else { + Some(extenddb_core::types::BillingMode::Provisioned) + }; + + // New backups preserve these fields. Keep the old 5/5 fallback + // for backups created before the metadata was added, while + // correctly omitting provisioned throughput for on-demand tables. + let mut provisioned_throughput: Option = + decode_optional(&backup_doc, "provisioned_throughput")?; + if let Some(throughput) = overrides.provisioned_throughput.clone() { + provisioned_throughput = Some(throughput); + } + let provisioned_throughput = + restore_provisioned_throughput(effective_billing, provisioned_throughput); + let global_secondary_indexes: Option> = + decode_optional(&backup_doc, "global_secondary_indexes")?; + let global_secondary_indexes = + restore_gsi_throughput(effective_billing, global_secondary_indexes); + let local_secondary_indexes: Option> = + decode_optional(&backup_doc, "local_secondary_indexes")?; + + // Preserve the source table's TableClass / SSESpecification / + // OnDemandThroughput settings when recreating. + let table_class = backup_doc.get_str("table_class").ok().map(str::to_owned); + let sse_specification: Option = + backup_doc.get("sse_specification").and_then(|b| { + if matches!(b, mongodb::bson::Bson::Null) { + None + } else { + bson::from_bson(b.clone()).ok() + } + }); + let on_demand_throughput: Option = + backup_doc.get("on_demand_throughput").and_then(|b| { + if matches!(b, mongodb::bson::Bson::Null) { + None + } else { + bson::from_bson(b.clone()).ok() + } + }); + + if effective_billing == "PROVISIONED" + && let Some(index) = global_secondary_indexes + .as_deref() + .unwrap_or_default() + .iter() + .find(|index| index.provisioned_throughput.is_none()) + { + return Err(StorageError::Validation(format!( + "One or more parameter values were invalid: GlobalSecondaryIndexOverride \ + must be specified for index: {} when BillingModeOverride is PROVISIONED", + index.index_name + ))); + } - let ks_json = serde_json::to_value(key_schema_bson) - .map_err(|e| StorageError::Internal(format!("serialize key_schema: {e}")))?; - let ad_json = serde_json::to_value(attr_defs_bson) - .map_err(|e| StorageError::Internal(format!("serialize attr_defs: {e}")))?; - - let key_schema: Vec = - serde_json::from_value(ks_json) - .map_err(|e| StorageError::Internal(format!("parse key_schema: {e}")))?; - let attr_defs: Vec = - serde_json::from_value(ad_json) - .map_err(|e| StorageError::Internal(format!("parse attr_defs: {e}")))?; - - let billing_mode = if billing == "PAY_PER_REQUEST" { - Some(extenddb_core::types::BillingMode::PayPerRequest) - } else { - Some(extenddb_core::types::BillingMode::Provisioned) + let create_input = extenddb_core::types::CreateTableInput { + table_name: target_table_name.clone(), + key_schema, + attribute_definitions: attr_defs, + billing_mode, + provisioned_throughput, + global_secondary_indexes, + local_secondary_indexes, + stream_specification: None, + tags: None, + deletion_protection_enabled: None, + sse_specification, + table_class, + on_demand_throughput, + // Fields for features this backend does not implement, vector + // indexes today, take their defaults. Adding one to + // CreateTableInput then does not break this build. + ..Default::default() + }; + Ok::<_, StorageError>(create_input) + } + .await; + let create_input = match prepared { + Ok(input) => input, + Err(error) => { + if let Err(cleanup) = restore_claim.release().await { + tracing::error!( + "restore of {backup_arn} failed validation ({error}) and its metadata \ + claim could not be cleared: {cleanup}" + ); + } + return Err(error); + } }; - // New backups preserve these fields. Keep the old 5/5 fallback - // for backups created before the metadata was added, while - // correctly omitting provisioned throughput for on-demand tables. - let provisioned_throughput: Option = - decode_optional(&backup_doc, "provisioned_throughput")?; - let provisioned_throughput = - restore_provisioned_throughput(billing, provisioned_throughput); - let global_secondary_indexes: Option> = - decode_optional(&backup_doc, "global_secondary_indexes")?; - let local_secondary_indexes: Option> = - decode_optional(&backup_doc, "local_secondary_indexes")?; - - // Preserve the source table's TableClass / SSESpecification / - // OnDemandThroughput settings when recreating. - let table_class = backup_doc.get_str("table_class").ok().map(str::to_owned); - let sse_specification: Option = - backup_doc.get("sse_specification").and_then(|b| { - if matches!(b, mongodb::bson::Bson::Null) { - None - } else { - bson::from_bson(b.clone()).ok() + // Provenance is part of the initial CREATING document. Keep the + // metadata marker until target creation returns: protection then + // overlaps, and DeleteBackup can observe neither an unclaimed + // backup nor an ownerless target. + let restore_at = bson::DateTime::now(); + let created = self + .create_table_for_restore( + &account_id, + create_input, + &backup_arn, + restore_at, + &restore_operation_id, + ) + .await; + let mut desc = match created { + Ok(desc) => { + if let Err(claim_error) = restore_claim.release().await { + if let Err(cleanup) = self + .abort_backup_restore( + &account_id, + &target_table_name, + &backup_arn, + &restore_operation_id, + ) + .await + { + tracing::error!( + "restore of {backup_arn} could not transfer its metadata claim \ + ({claim_error}) and target cleanup failed: {cleanup}" + ); + } + return Err(claim_error); } - }); - let on_demand_throughput: Option = - backup_doc.get("on_demand_throughput").and_then(|b| { - if matches!(b, mongodb::bson::Bson::Null) { - None - } else { - bson::from_bson(b.clone()).ok() + desc + } + Err(error) => { + if let Err(cleanup) = self + .abort_backup_restore( + &account_id, + &target_table_name, + &backup_arn, + &restore_operation_id, + ) + .await + { + tracing::error!( + "restore of {backup_arn} into {target_table_name} failed during \ + target creation ({error}), and cleanup failed: {cleanup}" + ); } - }); - - let create_input = extenddb_core::types::CreateTableInput { - table_name: target_table_name.clone(), - key_schema, - attribute_definitions: attr_defs, - billing_mode, - provisioned_throughput, - global_secondary_indexes, - local_secondary_indexes, - stream_specification: None, - tags: None, - deletion_protection_enabled: None, - sse_specification, - table_class, - on_demand_throughput, - // Fields for features this backend does not implement, vector - // indexes today, take their defaults. Adding one to - // CreateTableInput then does not break this build. - ..Default::default() + if let Err(cleanup) = restore_claim.release().await { + tracing::error!( + "restore of {backup_arn} failed during target creation ({error}), and \ + its metadata claim could not be cleared: {cleanup}" + ); + } + return Err(error); + } }; + #[allow(clippy::cast_precision_loss)] + { + desc.restore_summary = Some(extenddb_core::types::RestoreSummary { + source_backup_arn: Some(backup_arn.clone()), + restore_date_time: restore_at.timestamp_millis() as f64 / 1000.0, + restore_in_progress: true, + }); + } - // Create the table with the ACTIVE transition deferred: it enters - // CREATING with no scheduled flip, so the table cannot become - // ACTIVE until we schedule it below, after the data copy completes. - let desc = self - .create_table_impl(&account_id, create_input, true) - .await?; - - // Restore items from the backup collection using server-side `$out`. - // The backup collection was written by `create_backup` in the same - // document shape as the source data collection, so this is a - // direct clone — no per-item transformation needed. - let backup_id = backup_doc - .get_str("backup_id") - .map_err(|_| { - StorageError::Internal("backup metadata missing backup_id".to_string()) - })? - .to_owned(); - let src_coll_name = backup_collection_name(&backup_id); - let src_coll = self.data_db.collection::(&src_coll_name); - let new_coll_name = data_collection_name(&desc.table_id); - - // The test-hook gate holds the restore after its CREATING index - // metadata exists but before the base `$out` copy begins. This - // lets the integration test force a worker tick through the - // dangerous pre-copy window. - self.wait_for_gsi_backfill_test_gate(&target_table_name) - .await?; - - let pipeline = vec![doc! { "$out": &new_coll_name }]; - let out_cursor = src_coll - .aggregate(pipeline) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - let _drained: Vec = out_cursor - .try_collect() - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - - let new_data_coll = self.data_db.collection::(&new_coll_name); - let item_count = new_data_coll - .count_documents(doc! {}) - .await - .map_err(|e| StorageError::Internal(e.to_string()))? - as i64; + let copied = async { + // Restore items from the backup collection using server-side + // `$out`. The backup collection has the source document shape, + // so this is a direct clone with no per-item transformation. + let backup_id = backup_doc + .get_str("backup_id") + .map_err(|_| { + StorageError::Internal("backup metadata missing backup_id".to_string()) + })? + .to_owned(); + let src_coll_name = backup_collection_name(&backup_id); + let src_coll = self.data_db.collection::(&src_coll_name); + let new_coll_name = data_collection_name(&desc.table_id); + + // The test-hook gate holds the restore after its CREATING index + // metadata exists but before the base `$out` copy begins. This + // lets the integration test force a worker tick through the + // dangerous pre-copy window. + self.wait_for_gsi_backfill_test_gate(&target_table_name) + .await?; + + let pipeline = vec![doc! { "$out": &new_coll_name }]; + let out_cursor = src_coll + .aggregate(pipeline) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + let _drained: Vec = out_cursor + .try_collect() + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; - // The base `$out` copy is complete, but restored secondary-index - // collections are still empty. Mark the table as pending restore - // completion and leave its indexes CREATING so the shared worker - // can backfill them in bounded, restartable batches. - let status_update = doc! { - "$set": { - "item_count": item_count, - "restore_backfill_pending": true, - }, - }; - let tables_coll = self.catalog_db.collection::("tables"); - tables_coll + let new_data_coll = self.data_db.collection::(&new_coll_name); + let item_count = new_data_coll + .count_documents(doc! {}) + .await + .map_err(|e| StorageError::Internal(e.to_string()))? + as i64; + + // The base `$out` copy is complete, but restored secondary-index + // collections are still empty. Mark the table as pending restore + // completion and leave its indexes CREATING so the shared worker + // can backfill them in bounded, restartable batches. + let status_update = doc! { + "$set": { + "item_count": item_count, + "restore_backfill_pending": true, + }, + "$unset": { "restore_operation_id": "" }, + }; + let tables_coll = self.catalog_db.collection::("tables"); + tables_coll .update_one( doc! { "_id": { "account_id": &account_id, "table_name": &target_table_name } }, status_update, @@ -702,10 +1140,38 @@ impl BackupEngine for MongoEngine { .await .map_err(|e| StorageError::Internal(e.to_string()))?; - // The table remains CREATING until the worker has populated every - // restored index. `desc` therefore reports CREATING, matching the - // DynamoDB restore lifecycle while allowing the request to return - // before potentially hundreds of thousands of index writes finish. + // The table remains CREATING until the worker has populated + // every restored index. The response therefore matches the + // restore lifecycle without waiting for the backfill. + Ok::<(), StorageError>(()) + } + .await; + + if let Err(e) = copied { + match self + .abort_backup_restore( + &account_id, + &target_table_name, + &backup_arn, + &restore_operation_id, + ) + .await + { + Ok(true) => tracing::error!( + "restore of {backup_arn} into {target_table_name} failed and the \ + partial target was removed: {e}" + ), + Ok(false) => tracing::error!( + "restore of {backup_arn} into {target_table_name} failed after its \ + target was already removed: {e}" + ), + Err(cleanup) => tracing::error!( + "restore of {backup_arn} into {target_table_name} failed ({e}), and \ + physical cleanup also failed: {cleanup}" + ), + } + return Err(e); + } Ok(desc) }) @@ -813,7 +1279,12 @@ impl BackupEngine for MongoEngine { .create_backup(&account_id, &source_table_name, "__pitr_restore__") .await?; let desc = self - .restore_table_from_backup(&account_id, &target_table_name, &backup.backup_arn) + .restore_table_from_backup( + &account_id, + &target_table_name, + &backup.backup_arn, + RestoreTableOverrides::default(), + ) .await?; let _ = self.delete_backup(&account_id, &backup.backup_arn).await; Ok(desc) @@ -823,10 +1294,40 @@ impl BackupEngine for MongoEngine { #[cfg(test)] mod tests { - use super::{insert_non_empty_array, restore_provisioned_throughput}; + use super::{ + delete_claim_filter, insert_non_empty_array, restore_claim_filter, restore_gsi_throughput, + restore_provisioned_throughput, restoring_target_filter, + }; use extenddb_core::types::{GsiInput, ProvisionedThroughput}; use mongodb::bson::Document; + #[test] + fn gsi_throughput_is_dropped_for_an_on_demand_table() { + use extenddb_core::types::{ + GsiInput, KeySchemaElement, KeyType, Projection, ProjectionType, + }; + let gsi = GsiInput { + index_name: "g".to_owned(), + key_schema: vec![KeySchemaElement { + attribute_name: "gpk".to_owned(), + key_type: KeyType::Hash, + }], + projection: Projection { + projection_type: ProjectionType::All, + non_key_attributes: None, + }, + provisioned_throughput: Some(ProvisionedThroughput { + read_capacity_units: 7, + write_capacity_units: 8, + }), + }; + let kept = restore_gsi_throughput("PROVISIONED", Some(vec![gsi.clone()])).expect("gsis"); + assert!(kept[0].provisioned_throughput.is_some()); + let dropped = restore_gsi_throughput("PAY_PER_REQUEST", Some(vec![gsi])).expect("gsis"); + assert!(dropped[0].provisioned_throughput.is_none()); + assert!(restore_gsi_throughput("PAY_PER_REQUEST", None).is_none()); + } + #[test] fn backup_omits_empty_secondary_index_metadata() { let mut backup = Document::new(); @@ -870,4 +1371,33 @@ mod tests { None ); } + + #[test] + fn restore_and_delete_claims_exclude_each_other() { + let restore = restore_claim_filter("123456789012", "backup"); + let delete = delete_claim_filter("123456789012", "backup"); + for filter in [&restore, &delete] { + assert_eq!(filter.get_str("backup_status"), Ok("AVAILABLE")); + assert_eq!( + filter + .get_document("restore_in_progress") + .and_then(|predicate| predicate.get_bool("$exists")), + Ok(false) + ); + assert_eq!( + filter + .get_document("delete_in_progress") + .and_then(|predicate| predicate.get_bool("$exists")), + Ok(false) + ); + } + } + + #[test] + fn durable_restore_target_predicate_is_account_scoped_and_creating() { + let filter = restoring_target_filter("123456789012", "backup"); + assert_eq!(filter.get_str("_id.account_id"), Ok("123456789012")); + assert_eq!(filter.get_str("restore_source_backup_arn"), Ok("backup")); + assert_eq!(filter.get_str("table_status"), Ok("CREATING")); + } } diff --git a/crates/storage-mongodb/src/lib.rs b/crates/storage-mongodb/src/lib.rs index fe57bb3e9..4aac5a5ae 100644 --- a/crates/storage-mongodb/src/lib.rs +++ b/crates/storage-mongodb/src/lib.rs @@ -124,6 +124,17 @@ fn server_components_factory( details: e.to_string(), })?; + // A crash can leave a backup's metadata claimed by a restore or a + // delete that will never finish; clear those before taking requests. + match engine.sweep_stale_backup_claims().await { + Ok((0, 0)) => {} + Ok((restores, deletes)) => tracing::warn!( + "cleared {restores} stale restore claim(s) and finished {deletes} interrupted \ + DeleteBackup(s) left by the last shutdown" + ), + Err(e) => tracing::error!("Failed to sweep stale backup claims: {e}"), + } + let engine = Arc::new(engine); // Create one shared catalog/management client. MongoDB Client clones diff --git a/crates/storage-mongodb/src/table_engine.rs b/crates/storage-mongodb/src/table_engine.rs index 87a5df208..2e0573ee8 100644 --- a/crates/storage-mongodb/src/table_engine.rs +++ b/crates/storage-mongodb/src/table_engine.rs @@ -145,6 +145,36 @@ impl MongoEngine { account_id: &str, input: CreateTableInput, defer_active: bool, + ) -> Result { + self.create_table_impl_inner(account_id, input, defer_active, None) + .await + } + + /// Create a restore target with provenance present in its first catalog + /// document, so backup deletion cannot observe an ownerless target. + pub(crate) async fn create_table_for_restore( + &self, + account_id: &str, + input: CreateTableInput, + backup_arn: &str, + restore_at: bson::DateTime, + restore_operation_id: &str, + ) -> Result { + self.create_table_impl_inner( + account_id, + input, + true, + Some((backup_arn, restore_at, restore_operation_id)), + ) + .await + } + + async fn create_table_impl_inner( + &self, + account_id: &str, + input: CreateTableInput, + defer_active: bool, + restore: Option<(&str, bson::DateTime, &str)>, ) -> Result { Self::validate_account_id(account_id)?; @@ -232,7 +262,7 @@ impl MongoEngine { ) }; - let table_doc = doc! { + let mut table_doc = doc! { "_id": { "account_id": account_id, "table_name": &input.table_name }, "key_schema": key_schema_bson, "attribute_definitions": attr_defs_bson, @@ -253,6 +283,11 @@ impl MongoEngine { "sse_specification": sse_spec_bson, "on_demand_throughput": on_demand_bson, }; + if let Some((backup_arn, restore_at, restore_operation_id)) = restore { + table_doc.insert("restore_source_backup_arn", backup_arn); + table_doc.insert("restore_date_time", restore_at); + table_doc.insert("restore_operation_id", restore_operation_id); + } let tables_coll = self.catalog_db.collection::("tables"); tables_coll.insert_one(table_doc).await.map_err(|e| { @@ -565,6 +600,17 @@ impl MongoEngine { return Err(StorageError::DeletionProtected(input.table_name.clone())); } + // Restore targets have no ordinary timed CREATING transition: the + // restore worker owns the name until the item and index copy completes. + // Deleting one mid-copy can leave orphaned physical collections, so + // match the SQL backends and DynamoDB by refusing it while CREATING. + if desc.table_status == TableStatus::Creating && desc.restore_summary.is_some() { + return Err(StorageError::IndexesInUse(format!( + "Attempt to change a resource which is still in use: Table is being restored: {}", + input.table_name + ))); + } + // Mark as DELETING let tables_coll = self.catalog_db.collection::("tables"); tables_coll @@ -742,12 +788,17 @@ impl MongoEngine { // Build update document let mut update_doc = Document::new(); + let clear_provisioned_throughput = + matches!(input.billing_mode, Some(BillingMode::PayPerRequest)); if let Some(billing_mode) = &input.billing_mode { let billing_str = match billing_mode { BillingMode::Provisioned => "PROVISIONED", BillingMode::PayPerRequest => "PAY_PER_REQUEST", }; update_doc.insert("billing_mode", billing_str); + if clear_provisioned_throughput { + update_doc.insert("provisioned_throughput", bson::Bson::Null); + } } if let Some(pt) = &input.provisioned_throughput { @@ -821,6 +872,27 @@ impl MongoEngine { .map_err(|e| StorageError::Internal(e.to_string()))?; } + if clear_provisioned_throughput { + let table_doc = tables_coll + .find_one(doc! { + "_id": { "account_id": account_id, "table_name": &input.table_name }, + }) + .await + .map_err(|e| StorageError::Internal(e.to_string()))? + .ok_or_else(|| StorageError::TableNotFound(input.table_name.clone()))?; + let table_id = table_doc + .get_str("table_id") + .map_err(|_| StorageError::Internal("missing table_id".to_owned()))?; + self.catalog_db + .collection::("indexes") + .update_many( + doc! { "_id.table_id": table_id, "index_type": "GSI" }, + doc! { "$set": { "provisioned_throughput": bson::Bson::Null } }, + ) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + } + // Handle GSI updates if let Some(gsi_updates) = &input.global_secondary_index_updates { // Persist the request's attribute definitions, merged into the stored @@ -1388,7 +1460,17 @@ impl MongoEngine { number_of_decreases_today: 0, last_increase_date_time: None, last_decrease_date_time: None, - }); + }) + // The service reports ProvisionedThroughput on every + // index, 0/0 for an on-demand table, and so does the + // PostgreSQL backend. + .or(Some(ProvisionedThroughputDescription { + read_capacity_units: 0, + write_capacity_units: 0, + number_of_decreases_today: 0, + last_increase_date_time: None, + last_decrease_date_time: None, + })); gsis.push(GsiDescription { index_name: idx_name, @@ -1456,6 +1538,20 @@ impl MongoEngine { let on_demand_throughput: Option = doc .get("on_demand_throughput") .and_then(|b| bson::from_bson(b.clone()).ok()); + // Set on a table created by RestoreTableFromBackup; in progress until + // the table is ACTIVE (the restore's index backfill runs while it is + // CREATING). + #[allow(clippy::cast_precision_loss)] + let restore_summary = doc.get_str("restore_source_backup_arn").ok().map(|arn| { + extenddb_core::types::RestoreSummary { + source_backup_arn: Some(arn.to_owned()), + restore_date_time: doc + .get_datetime("restore_date_time") + .map(|d| d.timestamp_millis() as f64 / 1000.0) + .unwrap_or(0.0), + restore_in_progress: table_status == TableStatus::Creating, + } + }); Ok(TableDescription { table_name, @@ -1479,6 +1575,7 @@ impl MongoEngine { sse_description, table_class_summary, on_demand_throughput, + restore_summary, // Fields for features this backend does not implement, vector // indexes today, take their defaults. Adding one to // TableDescription then does not break this build. @@ -1557,7 +1654,10 @@ impl MongoEngine { /// Drop a physical collection, treating an already-missing namespace as /// successful so lifecycle retries can continue their cleanup. - async fn drop_collection_if_exists(&self, coll_name: &str) -> Result<(), StorageError> { + pub(crate) async fn drop_collection_if_exists( + &self, + coll_name: &str, + ) -> Result<(), StorageError> { match self.data_db.collection::(coll_name).drop().await { Ok(()) => {} Err(e) if matches!(*e.kind, mongodb::error::ErrorKind::Command(ref c) if c.code == 26) => @@ -1570,7 +1670,10 @@ impl MongoEngine { /// Drop all physical secondary-index collections for a table before its /// catalog index documents are removed. - async fn drop_index_collections_for_table(&self, table_id: &str) -> Result<(), StorageError> { + pub(crate) async fn drop_index_collections_for_table( + &self, + table_id: &str, + ) -> Result<(), StorageError> { use futures::TryStreamExt; let indexes_coll = self.catalog_db.collection::("indexes"); diff --git a/crates/storage-postgres/migrations/003_backup_definitions.sql b/crates/storage-postgres/migrations/003_backup_definitions.sql new file mode 100644 index 000000000..0756c3cc2 --- /dev/null +++ b/crates/storage-postgres/migrations/003_backup_definitions.sql @@ -0,0 +1,41 @@ +-- Copyright 2026 ExtendDB contributors +-- SPDX-License-Identifier: Apache-2.0 +-- Migration 003: record a backup's table definition, and which backup a +-- restored table came from (catalog version 0.0.4). +-- +-- A backup kept the source table's key schema, attribute definitions, and +-- billing mode, and nothing else, so a restored table came back without its +-- global and local secondary indexes, with 5/5 provisioned throughput, and +-- without its table class or encryption settings. One row per backup holds +-- those, in the wire's own shape behind a version marker (see +-- `extenddb_storage::backup_definition`). A backup taken before this +-- migration has no row and restores as before. +-- +-- Written to tolerate a replay, like 002: the runner applies a migration and +-- records it in `schema_history` as two separate commits, so a crash in +-- between leaves this file applied but unrecorded, and the next migrate runs +-- it again. CREATE TABLE IF NOT EXISTS and the version UPDATE are idempotent. + +BEGIN; + +CREATE TABLE IF NOT EXISTS backup_definitions ( + backup_arn TEXT PRIMARY KEY REFERENCES backups(backup_arn) ON DELETE CASCADE, + definition JSONB NOT NULL +); + +-- One row per table created by RestoreTableFromBackup: the backup it came +-- from and when, reported by DescribeTable as RestoreSummary, and used to +-- refuse DeleteBackup while the restore is still running. Not a foreign key +-- to backups: the backup may be deleted after the restore, and the summary +-- keeps naming it, as on the service. +CREATE TABLE IF NOT EXISTS table_restores ( + table_id TEXT PRIMARY KEY REFERENCES tables(table_id) ON DELETE CASCADE, + source_backup_arn TEXT NOT NULL, + restore_date_time TIMESTAMPTZ NOT NULL DEFAULT NOW() +); + +CREATE INDEX IF NOT EXISTS idx_table_restores_backup ON table_restores (source_backup_arn); + +UPDATE settings SET value = '0.0.4' WHERE key = 'catalog_version'; + +COMMIT; diff --git a/crates/storage-postgres/src/backup_engine.rs b/crates/storage-postgres/src/backup_engine.rs index a93239850..d7c55f361 100755 --- a/crates/storage-postgres/src/backup_engine.rs +++ b/crates/storage-postgres/src/backup_engine.rs @@ -4,15 +4,24 @@ //! Backup and point-in-time recovery implementation for `PostgreSQL` storage. use extenddb_core::types::{ - BackupDescription, BackupDetails, BackupSummary, ContinuousBackupsDescription, - PointInTimeRecoveryDescription, SourceTableDetails, TableDescription, + BackupDescription, BackupDetails, BackupSummary, ContinuousBackupsDescription, GsiInput, Item, + KeySchemaElement, LsiInput, PointInTimeRecoveryDescription, Projection, ScalarAttributeType, + SourceTableDetails, TableDescription, +}; +use extenddb_storage::backup_definition::{ + BACKUP_DEFINITION_VERSION, BackupTableDefinition, COPY_BATCH_BYTES, COPY_BATCH_ITEMS, + ensure_single_part_base_key, throughput_from_catalog, }; -use extenddb_storage::BackupEngine; use extenddb_storage::error::StorageError; +use extenddb_storage::util::{SortKeyValue, composite_pk_to_text, parse_sk, sk_column_n}; +use extenddb_storage::{BackupEngine, RestoreTableOverrides}; use futures::future::BoxFuture; use crate::PostgresEngine; -use crate::data::data_table_name; +use crate::data::{ + all_sort_key_info, data_table_name, index_table_name, insert_index_row_multi, + item_has_index_keys, project_item_for_index, +}; /// Current epoch milliseconds, used as the leading component of a backup id. fn epoch_millis() -> u128 { @@ -41,6 +50,535 @@ fn pg_timestamp_to_epoch(ts: time::OffsetDateTime) -> f64 { ts.unix_timestamp() as f64 } +/// Seed for the 64-bit restore lock key. Restore locks use the single-`bigint` +/// advisory lock form, which PostgreSQL keeps in a key space separate from the +/// two-`int4` form the migration and vector-build locks use, so they cannot +/// meet; the seed keeps this key space distinct from any other `bigint` user. +const RESTORE_LOCK_SEED: i64 = 0x0045_4452; // 'E', 'D', 'R' + +/// Restores in flight at once in this process. Each holds one dedicated +/// connection for its lock on top of its pooled ones, so this bounds the +/// connections restores can open outside the configured pools. A restore +/// beyond the limit waits for a slot. +static RESTORE_SLOTS: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(4); + +/// How long a restore waits for a slot before it is refused with +/// `LimitExceededException`. +const RESTORE_SLOT_WAIT_SECS: u64 = 30; + +/// How old an unowned restore target must be before it is treated as +/// abandoned. The owner takes its lock milliseconds after creating the +/// target, so this only has to cover that gap with a wide margin. +const ABANDONED_RESTORE_GRACE_SECS: i32 = 60; + +/// A secondary index of a restore target, as the copy needs it. +struct RestoreIndex { + table: String, + key_schema: Vec, + projection: Projection, +} + +/// Session-scoped ownership of one restore, held for the life of the copy. +/// +/// A `pg_try_advisory_lock` on a dedicated connection: it dies with the +/// connection, so a crashed process's claim disappears on its own and the +/// abandoned-restore sweep can tell a dead restore from a running one. +/// +/// If only this connection is lost while the copy carries on (its backend is +/// terminated, say), the sweep may claim the target after the grace period. +/// That cannot leave a wrong table: the claim and the restore's flip to +/// ACTIVE are both conditional on CREATING, so exactly one wins, and the +/// restore then fails with "deleted while it was being restored". +struct RestoreOwner { + _conn: sqlx::PgConnection, + _slot: Option>, +} + +/// The advisory-lock key for a restore into `table_name` in `account_id`. +/// +/// Keyed by name rather than table id so the restore can hold it before the +/// target exists: the sweep must never see an unowned target, including in +/// the window while `create_table_impl` is still running DDL. The key is a +/// 64-bit `hashtextextended`, so a collision is negligible, and one would +/// only make the sweep skip a table while the other name's lock is held, +/// never remove a table it should not. +fn restore_lock_key(account_id: &str, table_name: &str) -> String { + format!("{account_id}/{table_name}") +} + +async fn try_restore_lock( + pool: &sqlx::PgPool, + account_id: &str, + table_name: &str, + slot: Option>, +) -> Result, StorageError> { + let options = pool.connect_options(); + let mut conn = ::connect_with(&options) + .await + .map_err(|e| StorageError::Internal(format!("restore lock connection: {e}")))?; + // This session sits idle for the whole copy. A server-side + // idle_session_timeout would end it and release the lock while the + // restore is still running, so disable it for this session when the + // server permits. Some managed PostgreSQL services reject this SET; + // ownership still works there, but remains subject to their idle timeout. + if let Err(e) = sqlx::query("SET idle_session_timeout = 0") + .execute(&mut conn) + .await + { + tracing::warn!( + "could not disable idle_session_timeout for the restore lock session; \ + continuing with the server setting: {e}" + ); + } + let taken: bool = sqlx::query_scalar("SELECT pg_try_advisory_lock(hashtextextended($1, $2))") + .bind(restore_lock_key(account_id, table_name)) + .bind(RESTORE_LOCK_SEED) + .fetch_one(&mut conn) + .await + .map_err(|e| StorageError::Internal(format!("restore lock: {e}")))?; + Ok(taken.then_some(RestoreOwner { + _conn: conn, + _slot: slot, + })) +} + +/// Insert one buffered batch of backup rows in one statement and clear the +/// buffers. Returns the number written. +async fn insert_backup_batch( + tx: &mut sqlx::Transaction<'_, sqlx::Postgres>, + backup_arn: &str, + pks: &mut Vec, + datas: &mut Vec, +) -> Result { + if pks.is_empty() { + return Ok(0); + } + let n = i64::try_from(pks.len()).unwrap_or(i64::MAX); + sqlx::query( + "INSERT INTO backup_items (backup_arn, pk, item_data) \ + SELECT $1, pk, item_data::jsonb FROM UNNEST($2::text[], $3::text[]) AS t(pk, item_data)", + ) + .bind(backup_arn) + .bind(&*pks) + .bind(&*datas) + .execute(&mut **tx) + .await + .map_err(db_err)?; + pks.clear(); + datas.clear(); + Ok(n) +} + +fn db_err(e: sqlx::Error) -> StorageError { + StorageError::Internal(format!("Database error: {e}")) +} + +impl PostgresEngine { + /// Read the parts of a table's definition a restore recreates. + async fn capture_table_definition( + conn: &mut sqlx::PgConnection, + table_id: &str, + ) -> Result<(BackupTableDefinition, Vec<(String, String)>), StorageError> { + let (billing_mode, pt, table_class, sse, on_demand): ( + String, + Option, + Option, + Option, + Option, + ) = sqlx::query_as( + "SELECT billing_mode, provisioned_throughput, table_class, sse_specification, \ + on_demand_throughput FROM tables WHERE table_id = $1", + ) + .bind(table_id) + .fetch_one(&mut *conn) + .await + .map_err(db_err)?; + + let rows: Vec<( + String, + String, + String, + serde_json::Value, + serde_json::Value, + Option, + )> = sqlx::query_as( + "SELECT index_name, index_id, index_type, key_schema, projection, provisioned_throughput \ + FROM indexes WHERE table_id = $1 ORDER BY index_name", + ) + .bind(table_id) + .fetch_all(&mut *conn) + .await + .map_err(db_err)?; + let mut gsis = Vec::new(); + let mut lsis = Vec::new(); + let mut gsi_ids = Vec::new(); + for (index_name, index_id, index_type, ks, proj, ipt) in rows { + let key_schema: Vec = serde_json::from_value(ks) + .map_err(|e| StorageError::Internal(format!("index key schema: {e}")))?; + let projection: Projection = serde_json::from_value(proj) + .map_err(|e| StorageError::Internal(format!("index projection: {e}")))?; + if index_type == "LSI" { + lsis.push(LsiInput { + index_name, + key_schema, + projection, + }); + } else { + gsi_ids.push((index_name.clone(), index_id)); + gsis.push(GsiInput { + index_name, + key_schema, + projection, + provisioned_throughput: throughput_from_catalog(ipt.as_ref()), + }); + } + } + + let vector_index_names: Vec = sqlx::query_scalar( + "SELECT index_name FROM vector_indexes WHERE table_id = $1 ORDER BY index_name", + ) + .bind(table_id) + .fetch_all(&mut *conn) + .await + .map_err(db_err)?; + + Ok(( + BackupTableDefinition { + version: BACKUP_DEFINITION_VERSION, + billing_mode, + provisioned_throughput: throughput_from_catalog(pt.as_ref()), + global_secondary_indexes: gsis, + local_secondary_indexes: lsis, + table_class, + sse_specification: sse, + on_demand_throughput: on_demand + .map(serde_json::from_value) + .transpose() + .map_err(|e| StorageError::Internal(format!("on-demand throughput: {e}")))?, + vector_index_names, + }, + gsi_ids, + )) + } + + /// Copy a backup's items into a freshly created restore target, populate + /// its secondary indexes, and mark it ACTIVE. + /// + /// Every key column is derived from the item itself, so the copy does not + /// depend on how the source table's columns were laid out. Rows are plain + /// INSERTs: two backup rows for one key are a corrupt backup and fail the + /// restore rather than silently collapsing into one item. Items are read + /// as a stream and written in batches, so memory is bounded by the batch + /// size; the whole copy is one data transaction, so a failure part-way + /// leaves the target empty. The target is CREATING throughout, which + /// refuses every data-plane request, so nothing else writes to it. + async fn copy_backup_items( + &self, + desc: &TableDescription, + backup_arn: &str, + ) -> Result<(), StorageError> { + use futures::TryStreamExt; + + let ddb_table = data_table_name(&desc.table_id); + let sort_keys = all_sort_key_info(&desc.key_schema, &desc.attribute_definitions); + let index_rows: Vec<(String, serde_json::Value, serde_json::Value)> = sqlx::query_as( + "SELECT index_id, key_schema, projection FROM indexes WHERE table_id = $1", + ) + .bind(&desc.table_id) + .fetch_all(&self.pool) + .await + .map_err(db_err)?; + let mut indexes = Vec::with_capacity(index_rows.len()); + for (index_id, ks, proj) in index_rows { + indexes.push(RestoreIndex { + table: index_table_name(&index_id), + key_schema: serde_json::from_value(ks) + .map_err(|e| StorageError::Internal(format!("index key schema: {e}")))?, + projection: serde_json::from_value(proj) + .map_err(|e| StorageError::Internal(format!("index projection: {e}")))?, + }); + } + + // The backup's rows are read from one catalog snapshot in which the + // backup is still AVAILABLE. DeleteBackup removes the rows and marks + // the backup DELETED in one transaction, so a concurrent delete is + // either wholly visible here (the restore fails) or not at all (the + // restore copies every row). + let mut snapshot = self + .pool + .begin_with("BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY") + .await + .map_err(db_err)?; + let available: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM backups WHERE backup_arn = $1 \ + AND backup_status = 'AVAILABLE')", + ) + .bind(backup_arn) + .fetch_one(&mut *snapshot) + .await + .map_err(db_err)?; + if !available { + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + } + let mut tx = self.data_pool.begin().await.map_err(db_err)?; + let mut rows = sqlx::query_scalar::<_, String>( + "SELECT item_data::text FROM backup_items WHERE backup_arn = $1", + ) + .bind(backup_arn) + .fetch(&mut *snapshot); + let mut batch: Vec<(Item, String)> = Vec::with_capacity(COPY_BATCH_ITEMS); + let mut batch_bytes: usize = 0; + let mut item_count: i64 = 0; + while let Some(item_json) = rows.try_next().await.map_err(db_err)? { + let item: Item = serde_json::from_str(&item_json) + .map_err(|e| StorageError::Internal(format!("Parse backup item: {e}")))?; + // The parsed item is kept for its keys and index projections; the + // budget counts its stored text, which bounds both forms. + batch_bytes += item_json.len(); + batch.push((item, item_json)); + if batch.len() == COPY_BATCH_ITEMS || batch_bytes >= COPY_BATCH_BYTES { + self.write_restore_batch( + &mut tx, &ddb_table, desc, &sort_keys, &indexes, &batch, backup_arn, + ) + .await?; + item_count += i64::try_from(batch.len()).unwrap_or(i64::MAX); + batch.clear(); + batch_bytes = 0; + } + } + drop(rows); + snapshot.commit().await.map_err(db_err)?; + if !batch.is_empty() { + self.write_restore_batch( + &mut tx, &ddb_table, desc, &sort_keys, &indexes, &batch, backup_arn, + ) + .await?; + item_count += i64::try_from(batch.len()).unwrap_or(i64::MAX); + } + tx.commit().await.map_err(db_err)?; + + let (table_size,): (i64,) = sqlx::query_as(&format!( + "SELECT COALESCE(pg_total_relation_size('{ddb_table}'), 0)" + )) + .fetch_one(&self.data_pool) + .await + .map_err(db_err)?; + + // Mark the restored table ACTIVE now that the copy has committed. The + // table was created with the transition deferred, so this is the first + // point it can become ACTIVE, by ordering rather than by timing. + // + // Only from CREATING. DeleteTable refuses a restore target, but the + // abandoned-restore sweep can claim one whose lock session was lost + // (moving it to DELETING); that claim wins, and flipping the row back + // to ACTIVE would leave a table whose data tables are being dropped. + let activated = sqlx::query( + "UPDATE tables SET item_count = $1, table_size_bytes = $2, table_status = 'ACTIVE', \ + status_transition_at = NULL WHERE table_id = $3 AND table_status = 'CREATING'", + ) + .bind(item_count) + .bind(table_size) + .bind(&desc.table_id) + .execute(&self.pool) + .await + .map_err(db_err)?; + if activated.rows_affected() == 0 { + // The target was claimed for removal while the copy ran; the + // copy's rows go with it. + return Err(StorageError::TableNotFound(format!( + "restore of {backup_arn} into {} did not complete: the table was deleted \ + while it was being restored", + desc.table_name + ))); + } + Ok(()) + } + + /// Write one batch of restored items: one multi-row INSERT into the base + /// table, then each item's row in every secondary index it belongs to. + #[allow(clippy::too_many_arguments)] + async fn write_restore_batch( + &self, + tx: &mut sqlx::Transaction<'_, sqlx::Postgres>, + ddb_table: &str, + desc: &TableDescription, + sort_keys: &[(&str, ScalarAttributeType)], + indexes: &[RestoreIndex], + batch: &[(Item, String)], + backup_arn: &str, + ) -> Result<(), StorageError> { + let mut cols = vec!["pk".to_owned()]; + cols.extend( + sort_keys + .iter() + .enumerate() + .map(|(i, &(_, t))| sk_column_n(i, t)), + ); + cols.push("item_data".to_owned()); + let width = cols.len(); + // The last column is the item, bound as text and cast in SQL. + let tuples: Vec = (0..batch.len()) + .map(|r| { + let ps: Vec = (1..=width) + .map(|c| { + if c == width { + format!("${}::jsonb", r * width + c) + } else { + format!("${}", r * width + c) + } + }) + .collect(); + format!("({})", ps.join(", ")) + }) + .collect(); + let sql = format!( + "INSERT INTO {ddb_table} ({}) VALUES {}", + cols.join(", "), + tuples.join(", ") + ); + let mut query = sqlx::query(&sql); + for (item, item_json) in batch { + query = query.bind(composite_pk_to_text(item, &desc.key_schema)?); + for &(name, sk_type) in sort_keys { + let value = item.get(name).ok_or_else(|| { + StorageError::Internal(format!( + "backup {backup_arn} has an item without sort key {name}" + )) + })?; + query = match parse_sk(value, sk_type)? { + SortKeyValue::S(v) => query.bind(v), + SortKeyValue::N(v) => query.bind(v), + SortKeyValue::B(v) => query.bind(v), + }; + } + query = query.bind(item_json.as_str()); + } + query.execute(&mut **tx).await.map_err(db_err)?; + + if indexes.is_empty() { + return Ok(()); + } + let base_sks = all_sort_key_info(&desc.key_schema, &desc.attribute_definitions); + for idx in indexes { + let idx_sks = all_sort_key_info(&idx.key_schema, &desc.attribute_definitions); + for (item, _) in batch { + if !item_has_index_keys(item, &idx.key_schema) { + continue; + } + let projected = project_item_for_index( + item, + &idx.key_schema, + &desc.key_schema, + &idx.projection, + ); + insert_index_row_multi( + tx, + &idx.table, + item, + &projected, + &idx.key_schema, + &desc.key_schema, + &idx_sks, + &base_sks, + ) + .await?; + } + } + Ok(()) + } + + /// Remove a restore target whose copy failed or was abandoned. + /// + /// Keyed by table id, not name, so it can only ever remove the table that + /// restore created, and synchronous regardless of the control-plane + /// delay: the ordinary DeleteTable path would leave the name held in + /// DELETING for that long. + /// + /// The target is first claimed by moving it from CREATING to DELETING in + /// one statement. That is what makes the removal and the restore's own + /// flip to ACTIVE mutually exclusive: both are conditional on CREATING, so + /// exactly one wins. The claim also schedules the row for the ordinary + /// DELETING removal, so if a drop below fails, the row stays DELETING with + /// its index rows and the control plane finishes the job later. + async fn abort_restore(&self, table_id: &str) -> Result { + let mut tx = self.pool.begin().await.map_err(db_err)?; + let claimed = sqlx::query( + "UPDATE tables SET table_status = 'DELETING', status_transition_at = NOW() \ + WHERE table_id = $1 AND table_status = 'CREATING' AND status_transition_at IS NULL", + ) + .bind(table_id) + .execute(&mut *tx) + .await + .map_err(db_err)? + .rows_affected() + > 0; + let index_ids: Vec = + sqlx::query_scalar("SELECT index_id FROM indexes WHERE table_id = $1") + .bind(table_id) + .fetch_all(&mut *tx) + .await + .map_err(db_err)?; + tx.commit().await.map_err(db_err)?; + if !claimed { + return Ok(false); + } + let mut drops = vec![data_table_name(table_id)]; + drops.extend(index_ids.iter().map(|id| index_table_name(id))); + for t in drops { + sqlx::query(&format!("DROP TABLE IF EXISTS {t}")) + .execute(&self.data_pool) + .await + .map_err(db_err)?; + } + sqlx::query("DELETE FROM tables WHERE table_id = $1 AND table_status = 'DELETING'") + .bind(table_id) + .execute(&self.pool) + .await + .map_err(db_err)?; + Ok(true) + } + + /// Remove restore targets whose restore died with its process. + /// + /// A restore target is the only table that sits CREATING with no + /// scheduled transition. One older than the grace period whose restore + /// lock is free has no live owner: the process that created it crashed or + /// was killed mid-copy, and nothing else would ever move it on. Removing + /// it frees the name; the client sees the table disappear, as it would + /// after a failed restore. A target whose lock is held is being copied by + /// a live process, here or on another instance, and is left alone. + /// + /// Returns the names of the tables removed. + pub(crate) async fn sweep_abandoned_restores(&self) -> Result, StorageError> { + let candidates: Vec<(String, String, String)> = sqlx::query_as( + "SELECT table_id, account_id, table_name FROM tables \ + WHERE table_status = 'CREATING' AND status_transition_at IS NULL \ + AND creation_date_time < NOW() - make_interval(secs => $1)", + ) + .bind(f64::from(ABANDONED_RESTORE_GRACE_SECS)) + .fetch_all(&self.pool) + .await + .map_err(db_err)?; + let mut removed = Vec::new(); + for (table_id, account_id, table_name) in candidates { + let Some(_owner) = try_restore_lock(&self.pool, &account_id, &table_name, None).await? + else { + continue; + }; + if self.abort_restore(&table_id).await? { + tracing::warn!( + "removed table {table_name} ({table_id}): its restore did not finish and \ + no process owns it" + ); + removed.push(table_name); + } + } + Ok(removed) + } +} + impl BackupEngine for PostgresEngine { fn create_backup( &self, @@ -52,6 +590,14 @@ impl BackupEngine for PostgresEngine { let table_name = table_name.to_string(); let backup_name = backup_name.to_string(); Box::pin(async move { + // The table's definition and its items must describe one instant. + // They live in different databases, so no single snapshot covers + // both. Instead the table row is held FOR SHARE from the first + // catalog read until the data snapshot has been taken: UpdateTable + // and DeleteTable both take it FOR UPDATE, so no definition change + // can commit in between, and the definition read here is the one + // in force when the data snapshot starts. + let mut meta = self.pool.begin().await.map_err(db_err)?; // Verify table exists and get metadata. let row: ( String, @@ -66,11 +612,12 @@ impl BackupEngine for PostgresEngine { "SELECT table_id, table_arn, key_schema, attribute_definitions, \ billing_mode, table_size_bytes, item_count, \ COALESCE(provisioned_throughput::text, '{}') \ - FROM tables WHERE account_id = $1 AND table_name = $2 AND table_status = 'ACTIVE'", + FROM tables WHERE account_id = $1 AND table_name = $2 AND table_status = 'ACTIVE' \ + FOR SHARE", ) .bind(&account_id) .bind(&table_name) - .fetch_optional(&self.pool) + .fetch_optional(&mut *meta) .await .map_err(|e| StorageError::Internal(format!("Database error: {e}")))? .ok_or_else(|| StorageError::TableNotFound(format!("Table not found: {table_name}")))?; @@ -79,9 +626,9 @@ impl BackupEngine for PostgresEngine { table_id, _table_arn, key_schema, - attr_defs, + mut attr_defs, billing_mode, - size_bytes, + _size_bytes, _item_count, _prov, ) = row; @@ -92,38 +639,6 @@ impl BackupEngine for PostgresEngine { id = backup_id() ); - // Snapshot items from the data table. - let ddb_table = data_table_name(&table_id); - let ddb_table_unquoted = ddb_table.trim_matches('"'); - let has_sk: bool = sqlx::query_scalar( - "SELECT EXISTS(SELECT 1 FROM information_schema.columns \ - WHERE table_name = $1 AND column_name = 'sk')", - ) - .bind(ddb_table_unquoted) - .fetch_one(&self.data_pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; - - let items: Vec<(String, Option, serde_json::Value)> = if has_sk { - sqlx::query_as(&format!("SELECT pk, sk, item_data FROM {ddb_table}")) - .fetch_all(&self.data_pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))? - } else { - sqlx::query_as::<_, (String, serde_json::Value)>(&format!( - "SELECT pk, item_data FROM {ddb_table}" - )) - .fetch_all(&self.data_pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))? - .into_iter() - .map(|(pk, data)| (pk, None, data)) - .collect() - }; - - #[allow(clippy::cast_possible_wrap)] - let actual_count = items.len() as i64; - // Snapshot the source table's vector index configuration alongside // its key schema. Restore refuses a backup whose snapshot is // non-empty rather than silently dropping a declared index, and it @@ -149,52 +664,179 @@ impl BackupEngine for PostgresEngine { FROM vector_indexes WHERE table_id = $1", ) .bind(&table_id) - .fetch_optional(&self.pool) + .fetch_optional(&mut *meta) .await .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; - // Wrap all catalog-side writes in a single transaction so a crash - // cannot leave a backup marked AVAILABLE with partial items. - let mut tx = self - .pool - .begin() + let (mut definition, gsi_ids) = + Self::capture_table_definition(&mut meta, &table_id).await?; + + // Begin the data transaction and hold ACCESS SHARE while the + // catalog barrier still blocks UpdateTable and DeleteTable. LOCK + // TABLE is a utility statement and does not establish a + // REPEATABLE READ snapshot, so the real relation SELECT must run + // before meta.commit(). This fixes the data snapshot at the same + // instant as the captured definition; the lock then keeps a DROP + // from removing the relation while its rows are read. The + // control-plane drop path uses a short lock timeout and retries a + // blocked DeleteTable rather than stalling its worker pass. + let ddb_table = data_table_name(&table_id); + let mut snapshot = self + .data_pool + .begin_with("BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY") .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; + .map_err(db_err)?; + sqlx::query(&format!("LOCK TABLE {ddb_table} IN ACCESS SHARE MODE")) + .execute(&mut *snapshot) + .await + .map_err(db_err)?; + let _: Option = sqlx::query_scalar(&format!("SELECT 1 FROM {ddb_table} LIMIT 1")) + .fetch_optional(&mut *snapshot) + .await + .map_err(db_err)?; + meta.commit().await.map_err(db_err)?; + + // UpdateTable commits a new index's catalog row before it creates + // and fills the index's data table, and removes the row again if + // that fails. An index whose data table is not there yet is + // therefore not part of the table as of this snapshot; leave it + // out rather than record an index that may never exist. + let gsi_ids_len = gsi_ids.len(); + // The ids were read under the catalog barrier, so this probes + // exactly the indexes the captured definition describes. The + // to_regclass lookup reads pg_class rather than the pinned data + // snapshot and can see later committed DDL; that remains consistent: + // UpdateTable creates and fills a GSI in one data transaction, and + // restore rebuilds its rows from the captured base items. + let mut present = Vec::with_capacity(definition.global_secondary_indexes.len()); + for (gsi, (name, index_id)) in std::mem::take(&mut definition.global_secondary_indexes) + .into_iter() + .zip(gsi_ids) + { + debug_assert_eq!(gsi.index_name, name); + let exists: bool = sqlx::query_scalar("SELECT to_regclass($1) IS NOT NULL") + .bind(index_table_name(&index_id)) + .fetch_one(&mut *snapshot) + .await + .map_err(db_err)?; + if exists { + present.push(gsi); + } + } + let omitted = present.len() < gsi_ids_len; + definition.global_secondary_indexes = present; + // Leaving an index out can leave an attribute definition no key + // uses, which CreateTable would refuse on the service; drop those. + if omitted { + let mut used: std::collections::HashSet = + serde_json::from_value::>(key_schema.clone()) + .map_err(|e| StorageError::Internal(format!("key schema: {e}")))? + .into_iter() + .map(|k| k.attribute_name) + .collect(); + for k in definition + .global_secondary_indexes + .iter() + .flat_map(|g| &g.key_schema) + .chain( + definition + .local_secondary_indexes + .iter() + .flat_map(|l| &l.key_schema), + ) + { + used.insert(k.attribute_name.clone()); + } + if let Some(defs) = attr_defs.as_array_mut() { + defs.retain(|d| { + d.get("AttributeName") + .and_then(serde_json::Value::as_str) + .is_some_and(|n| used.contains(n)) + }); + } + } + let definition = definition.to_json()?; + + // All catalog-side writes are one transaction, so a crash cannot + // leave a backup AVAILABLE with partial items. The items are read + // from one REPEATABLE READ snapshot of the data table, streamed, + // and written in batches, so the backup is consistent as of one + // instant and memory is bounded by the batch size. + let mut tx = self.pool.begin().await.map_err(db_err)?; sqlx::query( "INSERT INTO backups (backup_arn, backup_name, table_id, table_name, account_id, \ backup_status, backup_size_bytes, item_count, key_schema, attribute_definitions, \ billing_mode, vector_indexes) \ - VALUES ($1, $2, $3, $4, $5, 'AVAILABLE', $6, $7, $8, $9, $10, $11)", + VALUES ($1, $2, $3, $4, $5, 'AVAILABLE', 0, 0, $6, $7, $8, $9)", ) .bind(&backup_arn) .bind(&backup_name) .bind(&table_id) .bind(&table_name) .bind(&account_id) - .bind(size_bytes) - .bind(actual_count) .bind(&key_schema) .bind(&attr_defs) .bind(&billing_mode) .bind(&vector_indexes) .execute(&mut *tx) .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; - - for (pk, sk, item_data) in &items { - sqlx::query( - "INSERT INTO backup_items (backup_arn, pk, sk, item_data) \ - VALUES ($1, $2, $3, $4)", - ) + .map_err(db_err)?; + sqlx::query("INSERT INTO backup_definitions (backup_arn, definition) VALUES ($1, $2)") .bind(&backup_arn) - .bind(pk) - .bind(sk.as_deref()) - .bind(item_data) + .bind(&definition) .execute(&mut *tx) .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; - } + .map_err(db_err)?; + + // `backup_items.sk` stays NULL: `item_data` is the whole item, key + // attributes included, and restore derives every key column from + // it. Reading the typed columns here instead would tie the backup + // to this table's physical layout. + let (actual_count, size_bytes) = { + use futures::TryStreamExt; + let mut count: i64 = 0; + let mut size: i64 = 0; + let mut pks: Vec = Vec::with_capacity(COPY_BATCH_ITEMS); + let mut datas: Vec = Vec::with_capacity(COPY_BATCH_ITEMS); + let mut batch_bytes: usize = 0; + // Rows come back as JSON text and are buffered as text, so the + // batch budget counts what is actually held. Each item is + // parsed only long enough to size it. + let select_sql = format!("SELECT pk, item_data::text FROM {ddb_table}"); + { + let mut rows = + sqlx::query_as::<_, (String, String)>(&select_sql).fetch(&mut *snapshot); + while let Some((pk, data)) = rows.try_next().await.map_err(db_err)? { + let item: Item = serde_json::from_str(&data) + .map_err(|e| StorageError::Internal(format!("Parse item: {e}")))?; + let item_bytes = extenddb_core::types::item_size_bytes(&item); + drop(item); + size += i64::try_from(item_bytes).unwrap_or(i64::MAX); + batch_bytes += data.len(); + pks.push(pk); + datas.push(data); + if pks.len() == COPY_BATCH_ITEMS || batch_bytes >= COPY_BATCH_BYTES { + batch_bytes = 0; + count += + insert_backup_batch(&mut tx, &backup_arn, &mut pks, &mut datas) + .await?; + } + } + } + count += insert_backup_batch(&mut tx, &backup_arn, &mut pks, &mut datas).await?; + snapshot.commit().await.map_err(db_err)?; + (count, size) + }; + sqlx::query( + "UPDATE backups SET item_count = $1, backup_size_bytes = $2 WHERE backup_arn = $3", + ) + .bind(actual_count) + .bind(size_bytes) + .bind(&backup_arn) + .execute(&mut *tx) + .await + .map_err(db_err)?; // Read back the creation timestamp assigned by the database. let created_at: time::OffsetDateTime = @@ -384,16 +1026,54 @@ impl BackupEngine for PostgresEngine { // reported missing here and the writes below never run. let desc = self.describe_backup(&account_id, &backup_arn).await?; - // The account predicate is repeated on both writes rather than - // relying on the lookup above, so the statements are correct on - // their own terms. + // One transaction, so a restore reading its snapshot sees either + // the whole backup or a DELETED one, never AVAILABLE with its rows + // gone. The account predicate is repeated on every write rather + // than relying on the lookup above, so the statements are correct + // on their own terms. + let mut tx = self.pool.begin().await.map_err(db_err)?; + // Refuse while a restore from this backup is still running, as the + // service does. The backup row is locked first; a restore records + // itself in `table_restores` while holding the same row FOR SHARE, + // so one of the two always sees the other. + sqlx::query( + "SELECT 1 FROM backups WHERE backup_arn = $1 AND account_id = $2 FOR UPDATE", + ) + .bind(&backup_arn) + .bind(&account_id) + .execute(&mut *tx) + .await + .map_err(db_err)?; + let restoring: Option = sqlx::query_scalar( + "SELECT t.table_name FROM table_restores r JOIN tables t ON t.table_id = r.table_id \ + WHERE r.source_backup_arn = $1 AND t.table_status = 'CREATING' LIMIT 1", + ) + .bind(&backup_arn) + .fetch_optional(&mut *tx) + .await + .map_err(db_err)?; + if let Some(table) = restoring { + return Err(StorageError::BackupInUse(format!( + "Backup is being used to restore table {table}: {backup_arn}" + ))); + } sqlx::query( "DELETE FROM backup_items WHERE backup_arn = $1 AND EXISTS (\ SELECT 1 FROM backups b WHERE b.backup_arn = $1 AND b.account_id = $2)", ) .bind(&backup_arn) .bind(&account_id) - .execute(&self.pool) + .execute(&mut *tx) + .await + .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; + + sqlx::query( + "DELETE FROM backup_definitions WHERE backup_arn = $1 AND EXISTS (\ + SELECT 1 FROM backups b WHERE b.backup_arn = $1 AND b.account_id = $2)", + ) + .bind(&backup_arn) + .bind(&account_id) + .execute(&mut *tx) .await .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; @@ -403,10 +1083,12 @@ impl BackupEngine for PostgresEngine { ) .bind(&backup_arn) .bind(&account_id) - .execute(&self.pool) + .execute(&mut *tx) .await .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; + tx.commit().await.map_err(db_err)?; + Ok(BackupDescription { backup_details: BackupDetails { backup_status: "DELETED".to_owned(), @@ -422,42 +1104,42 @@ impl BackupEngine for PostgresEngine { account_id: &str, target_table_name: &str, backup_arn: &str, + overrides: RestoreTableOverrides, ) -> BoxFuture<'_, Result> { let account_id = account_id.to_string(); let target_table_name = target_table_name.to_string(); let backup_arn = backup_arn.to_string(); Box::pin(async move { + #[allow(clippy::type_complexity)] let backup_row: ( - String, serde_json::Value, serde_json::Value, String, Option, Option, ) = sqlx::query_as( - "SELECT table_name, key_schema, attribute_definitions, billing_mode, \ - provisioned_throughput, vector_indexes \ - FROM backups \ - WHERE backup_arn = $1 AND account_id = $2 AND backup_status = 'AVAILABLE'", + "SELECT b.key_schema, b.attribute_definitions, b.billing_mode, \ + b.vector_indexes, d.definition \ + FROM backups b LEFT JOIN backup_definitions d ON d.backup_arn = b.backup_arn \ + WHERE b.backup_arn = $1 AND b.account_id = $2 AND b.backup_status = 'AVAILABLE'", ) .bind(&backup_arn) .bind(&account_id) .fetch_optional(&self.pool) .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))? + .map_err(db_err)? .ok_or_else(|| StorageError::Validation(format!("Backup not found: {backup_arn}")))?; - let (_orig_table, ks_json, ad_json, billing, _prov, vector_indexes) = backup_row; + let (ks_json, ad_json, billing, vector_indexes, definition) = backup_row; - // Restore rebuilds the target from the snapshot's key schema, - // attribute definitions and billing mode, and does not carry indexes - // across. For a source table that had vector indexes that would be - // silent loss of a declared index: the restored table would answer - // every request except a search, and the client would only find out - // on the first one. The service preserves vector indexes through + // Vector indexes are not restored on this backend. For a source + // table that had them, a restore without them would be silent loss + // of a declared index: the restored table would answer every + // request except a search, and the client would only find out on + // the first one. The service preserves vector indexes through // backup and restore (measured 2026-08-19), so the conformant end - // state is to restore them; until then this refuses, because a typed - // refusal is recoverable and silent loss is not. + // state is to restore them; until then this refuses, because a + // typed refusal is recoverable and silent loss is not. // // A NULL snapshot is a backup taken before the column existed, which // cannot have carried vector indexes: this backend could not create @@ -469,14 +1151,6 @@ impl BackupEngine for PostgresEngine { Some(snapshot) => { let version = snapshot.get("Version").and_then(serde_json::Value::as_u64); if version != Some(1) { - // Version skew, not a fault: a newer build wrote a shape - // this one does not know. Reported the same way as the - // refusal below, so a client gets a reason rather than a - // 500 and the operator gets no error-level noise. Nearly - // unreachable, because a shape change would normally come - // with a catalog version bump that the startup gate - // refuses first; the gap is a future build adding a member - // without any schema change. return Err(StorageError::Unsupported(format!( "backup {backup_arn} carries a vector index snapshot this build \ cannot read (version {version:?})" @@ -495,122 +1169,124 @@ impl BackupEngine for PostgresEngine { storage backend" ))); } + let definition = definition + .map(|d| BackupTableDefinition::from_json(d, &backup_arn)) + .transpose()?; + if let Some(d) = &definition { + d.ensure_restorable(&backup_arn)?; + } - let key_schema: Vec = - serde_json::from_value(ks_json) - .map_err(|e| StorageError::Internal(format!("Parse key schema: {e}")))?; + let key_schema: Vec = serde_json::from_value(ks_json) + .map_err(|e| StorageError::Internal(format!("Parse key schema: {e}")))?; + ensure_single_part_base_key(&key_schema, &backup_arn)?; let attr_defs: Vec = serde_json::from_value(ad_json) .map_err(|e| StorageError::Internal(format!("Parse attr defs: {e}")))?; - let billing_mode = if billing == "PAY_PER_REQUEST" { - Some(extenddb_core::types::BillingMode::PayPerRequest) - } else { - Some(extenddb_core::types::BillingMode::Provisioned) - }; - - let create_input = extenddb_core::types::CreateTableInput { + let mut create_input = extenddb_core::types::CreateTableInput { table_name: target_table_name.clone(), key_schema, attribute_definitions: attr_defs, - billing_mode, - provisioned_throughput: Some(extenddb_core::types::ProvisionedThroughput { - read_capacity_units: 5, - write_capacity_units: 5, - }), - global_secondary_indexes: None, - local_secondary_indexes: None, - stream_specification: None, - tags: None, - deletion_protection_enabled: None, - sse_specification: None, - table_class: None, - on_demand_throughput: None, ..Default::default() }; + match definition { + Some(d) => d.apply_to(&mut create_input, &overrides)?, + None => { + // A backup taken before table definitions were recorded + // carries only its keys and billing mode, so it restores as + // it always has: no secondary indexes, and 5/5 throughput for + // a provisioned table since the source's is unknown. + let on_demand = billing == "PAY_PER_REQUEST"; + create_input.billing_mode = Some(if on_demand { + extenddb_core::types::BillingMode::PayPerRequest + } else { + extenddb_core::types::BillingMode::Provisioned + }); + create_input.provisioned_throughput = + (!on_demand).then_some(extenddb_core::types::ProvisionedThroughput { + read_capacity_units: 5, + write_capacity_units: 5, + }); + overrides.apply_to_create_input(&mut create_input); + } + } - // Create the table with the ACTIVE transition deferred: it enters - // CREATING with no scheduled flip, so the control-plane worker - // cannot mark it ACTIVE while the item copy below is still - // running. The explicit ACTIVE update after the copy is the only - // way this table leaves CREATING, so ACTIVE implies the restored - // data is fully present. - let desc = self - .create_table_impl(&account_id, create_input, true) - .await?; - - let new_table_id = &desc.table_id; - let ddb_table = data_table_name(new_table_id); - let ddb_table_unquoted = ddb_table.trim_matches('"'); - - let has_sk: bool = sqlx::query_scalar( - "SELECT EXISTS(SELECT 1 FROM information_schema.columns \ - WHERE table_name = $1 AND column_name = 'sk')", + // Ownership first, before the target exists, so the + // abandoned-restore sweep never sees this target without an owner, + // including while the DDL below runs. It is held until the copy + // ends; a crash anywhere releases it with the connection, and the + // sweep then removes the target. A lock already held means another + // restore into this name is in flight right now; let it report the + // name as taken, as CreateTable would. + let slot = match tokio::time::timeout( + std::time::Duration::from_secs(RESTORE_SLOT_WAIT_SECS), + RESTORE_SLOTS.acquire(), ) - .bind(ddb_table_unquoted) - .fetch_one(&self.data_pool) .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; + { + Ok(slot) => { + slot.map_err(|e| StorageError::Internal(format!("restore slot: {e}")))? + } + Err(_) => { + return Err(StorageError::LimitExceeded( + "Too many restores are in progress on this server; retry later".to_owned(), + )); + } + }; + let Some(_owner) = + try_restore_lock(&self.pool, &account_id, &target_table_name, Some(slot)).await? + else { + return Err(StorageError::TableAlreadyExists(target_table_name.clone())); + }; - let items: Vec<(String, Option, serde_json::Value)> = - sqlx::query_as("SELECT pk, sk, item_data FROM backup_items WHERE backup_arn = $1") - .bind(&backup_arn) - .fetch_all(&self.pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; + // Create the table and register its source backup in one catalog + // transaction. The transaction holds the AVAILABLE backup row FOR + // SHARE until the CREATING target and its table_restores row commit, + // so DeleteBackup can neither remove the source between those rows + // nor observe an unregistered target. + let desc = self + .create_table_for_restore(&account_id, create_input, &backup_arn) + .await?; - for (pk, sk, item_data) in &items { - if has_sk { - sqlx::query(&format!( - "INSERT INTO {ddb_table} (pk, sk, item_data) VALUES ($1, $2, $3)" - )) - .bind(pk) - .bind(sk.as_deref()) - .bind(item_data) - .execute(&self.data_pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; - } else { - sqlx::query(&format!( - "INSERT INTO {ddb_table} (pk, item_data) VALUES ($1, $2)" - )) - .bind(pk) - .bind(item_data) - .execute(&self.data_pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; + // From here on a failure must not leave the target behind: it is + // CREATING with no scheduled transition, so nothing else would + // move it on, and the name would stay taken. Deleting the target + // cascades its table_restores row. + let copied = self.copy_backup_items(&desc, &backup_arn).await; + if let Err(e) = copied { + match self.abort_restore(&desc.table_id).await { + // Not ours to remove: it was already claimed for removal + // while it was being copied into, and that is why the copy + // failed. Report it as such rather than as a server fault a + // client would retry. + Ok(false) => { + return Err(StorageError::TableNotFound(format!( + "restore of {backup_arn} into {target_table_name} did not \ + complete: the table was deleted while it was being restored" + ))); + } + Ok(true) => tracing::error!( + "restore of {backup_arn} into {target_table_name} failed and the \ + partial table was removed: {e}" + ), + Err(cleanup) => tracing::error!( + "restore of {backup_arn} into {target_table_name} failed ({e}), and \ + the partial table ({}) could not be removed yet; the control plane \ + retries: {cleanup}", + desc.table_id + ), } + return Err(e); } - #[allow(clippy::cast_possible_wrap)] - let item_count = items.len() as i64; - - sqlx::query( - "UPDATE tables SET item_count = $1 WHERE account_id = $2 AND table_name = $3", - ) - .bind(item_count) - .bind(&account_id) - .bind(&target_table_name) - .execute(&self.pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; - - // Mark the restored table ACTIVE now that the copy has fully - // drained. The table was created with the transition deferred, so - // this is the first point it can become ACTIVE, by ordering rather - // than by timing. - sqlx::query( - "UPDATE tables SET table_status = 'ACTIVE', status_transition_at = NULL \ - WHERE account_id = $1 AND table_name = $2", - ) - .bind(&account_id) - .bind(&target_table_name) - .execute(&self.pool) - .await - .map_err(|e| StorageError::Internal(format!("Database error: {e}")))?; - // Return CREATING — the API response shows the initial status, // but the table is already ACTIVE by the time the caller polls. + // The summary carries the recorded restore time, so the response + // and later DescribeTable calls agree. + let mut desc = desc; + desc.restore_summary = self + .restore_summary(&desc.table_id, &desc.table_status) + .await?; Ok(desc) }) } @@ -733,7 +1409,12 @@ impl BackupEngine for PostgresEngine { .create_backup(&account_id, &source_table_name, "__pitr_restore__") .await?; let desc = self - .restore_table_from_backup(&account_id, &target_table_name, &backup.backup_arn) + .restore_table_from_backup( + &account_id, + &target_table_name, + &backup.backup_arn, + RestoreTableOverrides::default(), + ) .await?; let _ = self.delete_backup(&account_id, &backup.backup_arn).await; Ok(desc) diff --git a/crates/storage-postgres/src/create_table.rs b/crates/storage-postgres/src/create_table.rs index a221a1918..fcd5b226b 100755 --- a/crates/storage-postgres/src/create_table.rs +++ b/crates/storage-postgres/src/create_table.rs @@ -25,6 +25,32 @@ impl PostgresEngine { account_id: &str, input: CreateTableInput, defer_active: bool, + ) -> Result { + self.create_table_impl_inner(account_id, input, defer_active, None) + .await + } + + /// Create a restore target while holding a share lock on its AVAILABLE + /// source backup, and record the provenance before the catalog transaction + /// commits. A concurrent DeleteBackup therefore orders wholly before this + /// transaction (and no target is created) or wholly after it (and sees a + /// CREATING restore that is using the backup). + pub(crate) async fn create_table_for_restore( + &self, + account_id: &str, + input: CreateTableInput, + backup_arn: &str, + ) -> Result { + self.create_table_impl_inner(account_id, input, true, Some(backup_arn)) + .await + } + + async fn create_table_impl_inner( + &self, + account_id: &str, + input: CreateTableInput, + defer_active: bool, + restore_source_backup_arn: Option<&str>, ) -> Result { Self::validate_account_id(account_id)?; let table_id = uuid::Uuid::new_v4().to_string(); @@ -66,6 +92,23 @@ impl PostgresEngine { .await .map_err(|e| StorageError::Internal(e.to_string()))?; + if let Some(backup_arn) = restore_source_backup_arn { + let available: Option = sqlx::query_scalar( + "SELECT 1 FROM backups WHERE backup_arn = $1 AND account_id = $2 \ + AND backup_status = 'AVAILABLE' FOR SHARE", + ) + .bind(backup_arn) + .bind(account_id) + .fetch_optional(&mut *tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + if available.is_none() { + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + } + } + // Insert table metadata, returning creation timestamp and actual status. // Use PG error code 23505 for robust duplicate detection instead of string matching. // H-5: Insert as CREATING with a scheduled transition to ACTIVE, @@ -119,6 +162,15 @@ impl PostgresEngine { _ => StorageError::Internal(e.to_string()), })?; + if let Some(backup_arn) = restore_source_backup_arn { + sqlx::query("INSERT INTO table_restores (table_id, source_backup_arn) VALUES ($1, $2)") + .bind(&table_id) + .bind(backup_arn) + .execute(&mut *tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + } + // Insert GSI metadata // F-1: Store full ProvisionedThroughputDescription (not the input // ProvisionedThroughput) so DescribeTable can deserialize it without diff --git a/crates/storage-postgres/src/data/vector_index.rs b/crates/storage-postgres/src/data/vector_index.rs index 6259160cb..bb2852aa3 100644 --- a/crates/storage-postgres/src/data/vector_index.rs +++ b/crates/storage-postgres/src/data/vector_index.rs @@ -710,9 +710,11 @@ impl extenddb_storage::vector_lifecycle::VectorIndexBuild for PostgresVectorBuil // the advisory lock, and no decision reads this column. What it answers at // three in the morning is "which process is building this index", which the // lock cannot be asked from another session. - sqlx::query( - "UPDATE vector_indexes SET backfilling = true, build_owner = $3, \ - build_heartbeat_at = NOW() WHERE table_id = $1 AND index_id = $2", + let updated = sqlx::query( + "UPDATE vector_indexes v SET backfilling = true, build_owner = $3, \ + build_heartbeat_at = NOW() WHERE v.table_id = $1 AND v.index_id = $2 \ + AND EXISTS (SELECT 1 FROM tables t WHERE t.table_id = v.table_id \ + AND t.table_status <> 'DELETING')", ) .bind(&self.table_id) .bind(&self.index_id) @@ -720,6 +722,26 @@ impl extenddb_storage::vector_lifecycle::VectorIndexBuild for PostgresVectorBuil .execute(&self.catalog) .await .map_err(|e| StorageError::Internal(e.to_string()))?; + if updated.rows_affected() == 0 { + // Recovery may have recreated the physical table after DeleteTable's + // worker collected its ids or dropped the old table. No catalog row + // (or a DELETING parent) means the rebuild must stop and remove the + // table it just created rather than publish an orphan. + let mut tx = self + .data + .begin() + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + crate::PostgresEngine::drop_vector_data_table(&mut tx, &self.index_id).await?; + tx.commit() + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + release_hold(&self.data, &self.table_id, &self.index_id).await?; + return Err(StorageError::Internal(format!( + "vector index {} or its parent table vanished while recovery reset its data table", + self.index_id + ))); + } Ok(()) } @@ -775,8 +797,13 @@ impl extenddb_storage::vector_lifecycle::VectorIndexBuild for PostgresVectorBuil async fn reset_data_table(&mut self) -> Result<(), StorageError> { // Reload first: a rebuild has no request to read the definition from, and - // the index may have been altered since the build that died. - self.load_meta().await?; + // the index may have been altered since the build that died. If deletion + // won after candidate selection, give back the hold recovery just took; + // no later phase exists to release it. + if let Err(e) = self.load_meta().await { + release_hold(&self.data, &self.table_id, &self.index_id).await?; + return Err(e); + } let mut tx = self .data .begin() @@ -1010,6 +1037,14 @@ pub async fn reconcile_incomplete_vector_indexes( /// Without this, a build that dies after its first batch leaves its index /// `CREATING` and its queue hold in place, so the table's whole index propagation /// stops, and the only exit is a restart. +const REBUILD_CANDIDATES_SQL: &str = "SELECT v.index_id, v.table_id, v.dimensions, t.key_schema, t.attribute_definitions \ + FROM vector_indexes v JOIN tables t ON t.table_id = v.table_id \ + WHERE v.index_status = 'CREATING' AND t.table_status <> 'DELETING' \ + AND ($1::float8 IS NULL \ + OR v.build_heartbeat_at IS NULL \ + OR v.build_heartbeat_at < NOW() - make_interval(secs => $1)) \ + ORDER BY v.index_name"; + pub async fn rebuild_stuck_vector_indexes( engine: &crate::PostgresEngine, stale_after: Option, @@ -1017,19 +1052,12 @@ pub async fn rebuild_stuck_vector_indexes( // A null heartbeat counts as stale: it means the build never reached its first // batch, so nothing is renewing it. let stale_seconds = stale_after.map(|d| d.as_secs_f64()); - let rows: Vec<(String, String, i32, serde_json::Value, serde_json::Value)> = sqlx::query_as( - "SELECT v.index_id, v.table_id, v.dimensions, t.key_schema, t.attribute_definitions \ - FROM vector_indexes v JOIN tables t ON t.table_id = v.table_id \ - WHERE v.index_status = 'CREATING' \ - AND ($1::float8 IS NULL \ - OR v.build_heartbeat_at IS NULL \ - OR v.build_heartbeat_at < NOW() - make_interval(secs => $1)) \ - ORDER BY v.index_name", - ) - .bind(stale_seconds) - .fetch_all(&engine.pool) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + let rows: Vec<(String, String, i32, serde_json::Value, serde_json::Value)> = + sqlx::query_as(REBUILD_CANDIDATES_SQL) + .bind(stale_seconds) + .fetch_all(&engine.pool) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; // Swept against the UNFILTERED set of building indexes, not against the rows // selected above. Mid-run those rows are filtered by staleness, so they are not @@ -1103,3 +1131,14 @@ pub async fn rebuild_stuck_vector_indexes( } Ok(rebuilt) } + +#[cfg(test)] +mod tests { + use super::REBUILD_CANDIDATES_SQL; + + #[test] + fn stale_recovery_candidates_exclude_deleting_parents() { + assert!(REBUILD_CANDIDATES_SQL.contains("JOIN tables t ON t.table_id = v.table_id")); + assert!(REBUILD_CANDIDATES_SQL.contains("t.table_status <> 'DELETING'")); + } +} diff --git a/crates/storage-postgres/src/delete_table.rs b/crates/storage-postgres/src/delete_table.rs index 5a55231fe..2f56c318a 100755 --- a/crates/storage-postgres/src/delete_table.rs +++ b/crates/storage-postgres/src/delete_table.rs @@ -47,6 +47,29 @@ impl PostgresEngine { return Err(StorageError::DeletionProtected(row.table_arn.clone())); } + // A restore target (CREATING with no scheduled transition) is being + // filled by RestoreTableFromBackup. The service refuses to delete a + // table it is still creating, and deleting this one would make the + // restore fail part-way; refuse until the restore finishes. An + // abandoned target is removed by the restore sweep, not here. + if row.table_status == "CREATING" { + let restoring: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM tables WHERE table_id = $1 \ + AND table_status = 'CREATING' AND status_transition_at IS NULL)", + ) + .bind(&row.table_id) + .fetch_one(&mut *tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + if restoring { + return Err(StorageError::IndexesInUse(format!( + "Attempt to change a resource which is still in use: Table is being \ + restored: {}", + row.table_name + ))); + } + } + // Fetch indexes for the response description. let index_rows: Vec = sqlx::query_as( r"SELECT index_name, index_id, index_type, key_schema, projection, diff --git a/crates/storage-postgres/src/lib.rs b/crates/storage-postgres/src/lib.rs index 0d7a51c0d..63646cc1f 100755 --- a/crates/storage-postgres/src/lib.rs +++ b/crates/storage-postgres/src/lib.rs @@ -135,7 +135,7 @@ use sqlx::postgres::PgPoolOptions; /// /// The tuple is the single source of truth. Use `CATALOG_VERSION.to_string()` /// wherever a string representation is needed. -pub const CATALOG_VERSION: CatalogVersion = CatalogVersion::new(0, 0, 3); +pub const CATALOG_VERSION: CatalogVersion = CatalogVersion::new(0, 0, 4); /// Minimum number of connections allowed per pool. /// diff --git a/crates/storage-postgres/src/migrations.rs b/crates/storage-postgres/src/migrations.rs index 45af16167..4859e88ca 100755 --- a/crates/storage-postgres/src/migrations.rs +++ b/crates/storage-postgres/src/migrations.rs @@ -16,8 +16,18 @@ pub(crate) const CATALOG_MIGRATIONS: &[(&str, &str)] = &[ "002_vector_indexes.sql", include_str!("../../storage-postgres/migrations/002_vector_indexes.sql"), ), + ( + "003_backup_definitions.sql", + include_str!("../../storage-postgres/migrations/003_backup_definitions.sql"), + ), ]; +/// The runner's convergence write, executed after the walk over +/// [`CATALOG_MIGRATIONS`]. Same statement shape as the files' own in-file +/// writes, parameterized on the compiled version. +pub(crate) const SET_CATALOG_VERSION_SQL: &str = + "UPDATE settings SET value = $1 WHERE key = 'catalog_version'"; + /// Run catalog migrations, skipping already-applied ones. pub(crate) async fn run_catalog_migrations(pool: &PgPool) -> OpResult<()> { println!("--- Running catalog migrations..."); @@ -41,10 +51,49 @@ pub(crate) async fn run_catalog_migrations(pool: &PgPool) -> OpResult<()> { // another migration lands. record_migration(pool, filename).await?; } + // The runner owns the final version write. Each file still writes the + // version it introduces, but a replay can re-apply an EARLIER file while a + // later one stays recorded and skipped: the re-applied file's in-file write + // then leaves the stored version behind the schema actually present, the + // startup gate refuses the catalog, and a second migrate finds nothing + // unrecorded and writes nothing, stranding the deployment. Converging on + // the compiled version after every completed walk closes that gap. The + // in-file writes stay until #221 moves version ownership into the runner + // entirely. + converge_catalog_version(pool).await?; println!(" Migrations applied."); Ok(()) } +/// Write the compiled catalog version, but never move the stored version +/// backwards: a catalog stamped by a newer binary stays stamped, so that +/// binary's startup gate keeps working and this one's keeps refusing it. +/// `extenddb migrate` refuses such a catalog before the walk; this is the +/// runner-level guard for every other caller. +async fn converge_catalog_version(pool: &PgPool) -> OpResult<()> { + use extenddb_core::version::CatalogVersion; + let stored: Option<(String,)> = + sqlx::query_as("SELECT value FROM settings WHERE key = 'catalog_version'") + .fetch_optional(pool) + .await + .map_err(|e| OpError::Internal(format!("Read catalog_version: {e}")))?; + if let Some((stored,)) = &stored + && let Ok(stored_v) = stored.parse::() + && stored_v > crate::CATALOG_VERSION + { + return Err(OpError::Internal(format!( + "catalog version {stored} is newer than this binary's {}; not downgrading it", + crate::CATALOG_VERSION + ))); + } + sqlx::query(SET_CATALOG_VERSION_SQL) + .bind(crate::CATALOG_VERSION.to_string()) + .execute(pool) + .await + .map_err(|e| OpError::Internal(format!("Write catalog_version: {e}")))?; + Ok(()) +} + /// Embedded data-database migration files, applied in order. Tracked in the /// data database's own `schema_history` table (a separate database from the /// catalog), so `extenddb migrate` applies exactly the pending migrations. @@ -311,10 +360,10 @@ mod tests { fn the_migration_count_and_the_catalog_version_agree() { assert_eq!( CATALOG_MIGRATIONS.len(), - 2, + 3, "a catalog migration was added or removed; update CATALOG_VERSION and this count" ); - assert_eq!(CATALOG_VERSION.to_string(), "0.0.3"); + assert_eq!(CATALOG_VERSION.to_string(), "0.0.4"); } /// The version the binary expects must be the version the schema writes. diff --git a/crates/storage-postgres/src/table_helpers.rs b/crates/storage-postgres/src/table_helpers.rs index d75049ba0..5bbf78373 100755 --- a/crates/storage-postgres/src/table_helpers.rs +++ b/crates/storage-postgres/src/table_helpers.rs @@ -242,7 +242,9 @@ impl PostgresEngine { .map_err(|e| StorageError::Internal(e.to_string()))?; let table_name_owned = row.table_name.clone(); + let table_id = row.table_id.clone(); let mut desc = self.build_table_description_from_row(account_id, row, index_rows)?; + desc.restore_summary = self.restore_summary(&table_id, &desc.table_status).await?; desc.vector_indexes = vector_index_descriptions( &self.region, account_id, @@ -435,4 +437,26 @@ impl PostgresEngine { ..Default::default() }) } + + /// The RestoreSummary of a table created by RestoreTableFromBackup, or + /// `None` for any other table. In progress until the table is ACTIVE. + pub(crate) async fn restore_summary( + &self, + table_id: &str, + status: &extenddb_core::types::TableStatus, + ) -> Result, StorageError> { + let row: Option<(String, f64)> = sqlx::query_as( + "SELECT source_backup_arn, EXTRACT(EPOCH FROM restore_date_time)::FLOAT8 \ + FROM table_restores WHERE table_id = $1", + ) + .bind(table_id) + .fetch_optional(&self.pool) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + Ok(row.map(|(arn, at)| extenddb_core::types::RestoreSummary { + source_backup_arn: Some(arn), + restore_date_time: at, + restore_in_progress: *status == extenddb_core::types::TableStatus::Creating, + })) + } } diff --git a/crates/storage-postgres/src/worker_store.rs b/crates/storage-postgres/src/worker_store.rs index 372fbf49d..52ff9ced1 100755 --- a/crates/storage-postgres/src/worker_store.rs +++ b/crates/storage-postgres/src/worker_store.rs @@ -57,22 +57,19 @@ impl PostgresEngine { // DELETING → remove row (with tags and data table cleanup). // - // P57 Bug 1 fix: Collect index names BEFORE deleting the table row. - // The `indexes` table has `ON DELETE CASCADE` referencing `tables`, - // so `DELETE FROM tables` immediately removes all index rows. The old - // code did DELETE first then SELECT on indexes — always got zero rows, - // orphaning all GSI data tables. - // - // Strategy: SELECT ... FOR UPDATE SKIP LOCKED to lock candidates, - // collect index names, then DELETE. All in one transaction. + // Lock candidate catalog rows while their index ids are collected. The + // data tables are dropped before the catalog row: if any data DDL + // fails, the transaction rolls back and the row remains DELETING for + // the next pass. Once the data transaction commits, deleting the table + // row cascades its index and stream rows without orphaning data tables. let mut tx = self .pool .begin() .await .map_err(|e| StorageError::Internal(e.to_string()))?; - let candidates: Vec<(String, String, String, String)> = sqlx::query_as( - r"SELECT account_id, table_name, table_arn, table_id FROM tables + let candidates: Vec<(String, String, String)> = sqlx::query_as( + r"SELECT table_name, table_arn, table_id FROM tables WHERE table_status = 'DELETING' AND status_transition_at <= NOW() FOR UPDATE SKIP LOCKED", ) @@ -80,15 +77,9 @@ impl PostgresEngine { .await .map_err(|e| StorageError::Internal(e.to_string()))?; - // Collect index ids while the rows still exist. Vector indexes are read - // here for the same reason and at the same moment as the secondary ones: - // their catalog rows cascade away with the table row, and each id names a - // data table that still has to be dropped. - let mut drop_info: Vec<(String, Vec, Vec)> = Vec::new(); - - for (_acct_id, name, arn, table_id) in &candidates { - let index_ids: Vec<(String,)> = - sqlx::query_as("SELECT index_id FROM indexes WHERE table_id = $1") + for (name, arn, table_id) in &candidates { + let index_ids: Vec = + sqlx::query_scalar("SELECT index_id FROM indexes WHERE table_id = $1") .bind(table_id) .fetch_all(&mut *tx) .await @@ -100,6 +91,61 @@ impl PostgresEngine { .await .map_err(|e| StorageError::Internal(e.to_string()))?; + let dropped = async { + let mut data_tx = self + .data_pool + .begin() + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + // CreateBackup holds ACCESS SHARE through its item read. Do not + // let a DROP waiting for that lock stall every transition in + // this worker pass; a timeout rolls this transaction back and + // leaves the catalog row DELETING for a later retry. + sqlx::query("SET LOCAL lock_timeout = '3s'") + .execute(&mut *data_tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + for idx_id in &index_ids { + Self::drop_index_data_table(&mut data_tx, idx_id).await?; + } + for idx_id in &vector_index_ids { + Self::drop_vector_data_table(&mut data_tx, idx_id).await?; + } + Self::drop_data_table(&mut data_tx, table_id).await?; + + // Drop any still-pending GSI propagation rows for this table in + // the same transaction that drops the tables, so workers don't + // waste a claim→deserialize→attempt→skip cycle on orphaned rows. + sqlx::query("DELETE FROM gsi_pending WHERE table_id = $1") + .bind(table_id) + .execute(&mut *data_tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + + // The table is gone, so no vector build can release this hold; + // leaving it would block later claims for the table id. + sqlx::query("DELETE FROM vector_index_holds WHERE table_id = $1") + .bind(table_id) + .execute(&mut *data_tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + + data_tx + .commit() + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + Ok::<(), StorageError>(()) + } + .await; + + if let Err(e) = dropped { + tracing::warn!( + "could not drop data tables for '{name}' ({table_id}); the table remains \ + DELETING and the control plane will retry: {e}" + ); + continue; + } + // Delete tags explicitly (not covered by CASCADE from tables). sqlx::query("DELETE FROM tags WHERE resource_arn = $1") .bind(arn) @@ -107,19 +153,14 @@ impl PostgresEngine { .await .map_err(|e| StorageError::Internal(e.to_string()))?; - // Now delete the table row. CASCADE removes indexes and streams. + // The successful data drop makes it safe to remove the catalog row. + // CASCADE removes indexes and streams. sqlx::query("DELETE FROM tables WHERE table_id = $1") .bind(table_id) .execute(&mut *tx) .await .map_err(|e| StorageError::Internal(e.to_string()))?; - drop_info.push(( - table_id.clone(), - index_ids.into_iter().map(|(n,)| n).collect(), - vector_index_ids, - )); - transitions.push((name.clone(), "DELETING → deleted")); } @@ -127,42 +168,22 @@ impl PostgresEngine { .await .map_err(|e| StorageError::Internal(e.to_string()))?; - // P54 Bug 1: Drop data tables on the data pool after catalog commit. - for (table_id, index_ids, vector_index_ids) in &drop_info { - let mut data_tx = self - .data_pool - .begin() - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - for idx_id in index_ids { - Self::drop_index_data_table(&mut data_tx, idx_id).await?; + // CREATING with no owner → removed. A restore whose process died + // mid-copy leaves its target CREATING with no scheduled transition. + // The poller is notification-driven with a 60-second idle timeout, so + // the 60-second grace means an idle abandoned target is removed about + // 60–120 seconds after ownership is lost. Every pass opens a fresh + // out-of-pool connection for each old candidate to probe its advisory + // lock; a still-owned restore closes that probe and is checked again on + // the next pass. Failures are logged and skipped, so they cannot hold + // up the transitions above. + match self.sweep_abandoned_restores().await { + Ok(names) => { + for name in names { + transitions.push((name, "abandoned restore → removed")); + } } - for idx_id in vector_index_ids { - Self::drop_vector_data_table(&mut data_tx, idx_id).await?; - } - Self::drop_data_table(&mut data_tx, table_id).await?; - - // Drop any still-pending GSI propagation rows for this table in the - // same transaction that drops the tables, so workers don't waste a - // claim→deserialize→attempt→skip cycle on now-orphaned rows. - sqlx::query("DELETE FROM gsi_pending WHERE table_id = $1") - .bind(table_id) - .execute(&mut *data_tx) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - - // And any vector build hold: the table is gone, so nothing will ever - // release it, and a stale hold blocks claims for that table id. - sqlx::query("DELETE FROM vector_index_holds WHERE table_id = $1") - .bind(table_id) - .execute(&mut *data_tx) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - - data_tx - .commit() - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + Err(e) => tracing::warn!("abandoned restore sweep failed: {e}"), } Ok(transitions) diff --git a/crates/storage-postgres/tests/backup_restore.rs b/crates/storage-postgres/tests/backup_restore.rs new file mode 100644 index 000000000..425e34f91 --- /dev/null +++ b/crates/storage-postgres/tests/backup_restore.rs @@ -0,0 +1,1693 @@ +// Copyright 2026 ExtendDB contributors +// SPDX-License-Identifier: Apache-2.0 +//! Storage-level tests for PostgreSQL backup restore. +//! +//! The wire suite (`tests/test_backup_restore_fidelity.py`) checks round trips +//! on every backend. Two cases are only reachable here, because they need a +//! `backup_items` row the current binary would not write: +//! +//! - a backup written before the sort-key fix, whose rows carry the item with +//! `sk` NULL, must restore with its sort keys (they are read from the item); +//! - a backup whose copy fails part-way must not leave the target table +//! behind in CREATING, where nothing would ever move it on and its name +//! would stay taken; +//! - a backup with two rows for one key must fail rather than collapse them; +//! - a backup of a multi-part key table (a preview gated by +//! `enable_multipart_keys`) must be refused, not restored wrongly; +//! - the abandoned-restore sweep, which needs a crash to reach. +//! +//! Each test builds a throwaway catalog and data database and drops them when +//! it ends, passing or failing. The tests run one at a time: each opens its +//! own pools, and running them all at once can exhaust the server's +//! connection limit. Requires `EXTENDDB_TEST_PG_CONNECTION_STRING` (a base URL with no database +//! component); without it every test reports a skip and passes. + +use extenddb_core::types::{ + AttributeDefinition, BillingMode, CreateTableInput, DescribeTableInput, Item, KeySchemaElement, + KeyType, ScalarAttributeType, TableKeyInfo, +}; +use extenddb_storage::error::StorageError; +use extenddb_storage::{BackupEngine, DataEngine, RestoreTableOverrides, TableEngine}; +use extenddb_storage_postgres::{PostgresConfig, PostgresEngine}; +use sqlx::PgPool; +use sqlx::postgres::PgPoolOptions; + +const ACCOUNT: &str = "123456789012"; +const REGION: &str = "us-east-1"; + +/// A catalog database and a separate data database, the layout `extenddb +/// init` creates. They must be separate here: the catalog schema carries a +/// legacy `stream_shards` table with a foreign key to `tables`, which would +/// shadow the data schema's table if both were applied to one database. +struct Scratch { + engine: PostgresEngine, + /// The data database. + db: PgPool, + catalog: PgPool, + admin: PgPool, + db_names: [String; 2], + _guard: DbGuard, + _serial: tokio::sync::MutexGuard<'static, ()>, +} + +/// Drops the scratch databases if the test panics, from the moment they are +/// created, including during setup. +struct DbGuard { + names: Vec, +} + +/// One test at a time; see the module comment. +static SERIAL: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(()); + +impl Scratch { + /// Close the pools and drop the scratch databases. Also run, on a fresh + /// runtime, if a test panics before calling it. + async fn cleanup(self) { + self.db.close().await; + self.catalog.close().await; + drop_databases(&self.admin, &self.db_names).await; + self.admin.close().await; + } +} + +async fn drop_databases(admin: &PgPool, names: &[String]) { + for name in names { + let _ = sqlx::query(&format!("DROP DATABASE IF EXISTS \"{name}\" WITH (FORCE)")) + .execute(admin) + .await; + } +} + +impl Drop for DbGuard { + fn drop(&mut self) { + if !std::thread::panicking() { + return; + } + // A failed test: drop its databases on a separate thread and runtime, + // since the test's own runtime is unwinding. + let (Some(base), names) = (base_conn(), self.names.clone()) else { + return; + }; + let _ = std::thread::spawn(move || { + let Ok(rt) = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + else { + return; + }; + rt.block_on(async { + if let Ok(admin) = PgPoolOptions::new() + .max_connections(1) + .connect(&format!("{base}/postgres")) + .await + { + drop_databases(&admin, &names).await; + admin.close().await; + } + }); + }) + .join(); + } +} + +fn base_conn() -> Option { + let conn = std::env::var("EXTENDDB_TEST_PG_CONNECTION_STRING").ok()?; + (!conn.trim().is_empty()).then(|| conn.trim_end_matches('/').to_owned()) +} + +async fn connect(url: &str) -> PgPool { + PgPoolOptions::new() + .max_connections(2) + .connect(url) + .await + .expect("connect to a scratch database") +} + +async fn apply(pool: &PgPool, migrations: &[&str]) { + for sql in migrations { + sqlx::raw_sql(sql) + .execute(pool) + .await + .expect("apply a shipped migration"); + } +} + +async fn scratch() -> Scratch { + let serial = SERIAL.lock().await; + let base = base_conn().expect("caller checks base_conn() first"); + let stem = format!("eddb_bkup_{}", uuid::Uuid::new_v4().simple())[..24].to_owned(); + let (catalog_name, data_name) = (format!("{stem}_c"), format!("{stem}_d")); + let admin = PgPoolOptions::new() + .max_connections(1) + .connect(&format!("{base}/postgres")) + .await + .expect("connect to the postgres maintenance database"); + let guard = DbGuard { + names: vec![catalog_name.clone(), data_name.clone()], + }; + for name in [&catalog_name, &data_name] { + sqlx::query(&format!("CREATE DATABASE \"{name}\"")) + .execute(&admin) + .await + .expect("create a scratch database"); + } + let catalog_url = format!("{base}/{catalog_name}"); + let data_url = format!("{base}/{data_name}"); + let catalog = connect(&catalog_url).await; + let db = connect(&data_url).await; + apply( + &catalog, + &[ + include_str!("../migrations/001_schema.sql"), + include_str!("../migrations/002_vector_indexes.sql"), + include_str!("../migrations/003_backup_definitions.sql"), + ], + ) + .await; + apply( + &db, + &[ + include_str!("../data_migrations/001_data_schema.sql"), + include_str!("../data_migrations/002_gsi_pending.sql"), + include_str!("../data_migrations/003_idempotency_account_scope.sql"), + include_str!("../data_migrations/004_vector_index_state.sql"), + ], + ) + .await; + sqlx::query("UPDATE settings SET value = '0' WHERE key = 'control_plane_delay_seconds'") + .execute(&catalog) + .await + .expect("pin the control-plane delay to zero"); + sqlx::query("UPDATE settings SET value = '0' WHERE key = 'index_propagation_delay_ms'") + .execute(&catalog) + .await + .expect("make index propagation synchronous, as no queue worker runs here"); + sqlx::query( + "INSERT INTO settings (key, value) VALUES ('data_database_connection_string', $1) \ + ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value", + ) + .bind(&data_url) + .execute(&catalog) + .await + .expect("point the catalog at the data database"); + sqlx::query("INSERT INTO accounts (account_id, account_name) VALUES ($1, $2)") + .bind(ACCOUNT) + .bind(format!("acct-{stem}")) + .execute(&catalog) + .await + .expect("seed the account row"); + let engine = PostgresEngine::new( + &PostgresConfig { + connection_string: catalog_url, + pool_size: 4, + max_item_size_bytes: 400_000, + }, + REGION, + ) + .await + .expect("open a PostgresEngine on the scratch databases"); + Scratch { + engine, + db, + catalog, + admin, + db_names: [catalog_name, data_name], + _guard: guard, + _serial: serial, + } +} + +fn composite_table(name: &str) -> CreateTableInput { + let key = |name: &str, key_type| KeySchemaElement { + attribute_name: name.to_owned(), + key_type, + }; + let attr = |name: &str, attribute_type| AttributeDefinition { + attribute_name: name.to_owned(), + attribute_type, + }; + CreateTableInput { + table_name: name.to_owned(), + key_schema: vec![key("pk", KeyType::Hash), key("sk", KeyType::Range)], + attribute_definitions: vec![ + attr("pk", ScalarAttributeType::S), + attr("sk", ScalarAttributeType::N), + ], + billing_mode: Some(BillingMode::PayPerRequest), + ..Default::default() + } +} + +/// A backup of an empty composite-key table, then `rows` written straight +/// into `backup_items` in the pre-fix shape: `sk` NULL, the item in +/// `item_data`. +async fn legacy_backup(s: &Scratch, source: &str, rows: &[serde_json::Value]) -> String { + legacy_backup_of(s, composite_table(source), rows).await +} + +async fn legacy_backup_of( + s: &Scratch, + source: CreateTableInput, + rows: &[serde_json::Value], +) -> String { + let source_name = source.table_name.clone(); + s.engine + .create_table(ACCOUNT, source) + .await + .expect("create the source table"); + let backup = s + .engine + .create_backup(ACCOUNT, &source_name, "legacy") + .await + .expect("back up the empty source"); + for item in rows { + sqlx::query( + "INSERT INTO backup_items (backup_arn, pk, sk, item_data) VALUES ($1, $2, NULL, $3)", + ) + .bind(&backup.backup_arn) + .bind(item["pk"]["S"].as_str().unwrap_or("x")) + .bind(item) + .execute(&s.catalog) + .await + .expect("write a legacy backup row"); + } + backup.backup_arn +} + +async fn table_status(s: &Scratch, name: &str) -> Option { + sqlx::query_scalar("SELECT table_status FROM tables WHERE account_id = $1 AND table_name = $2") + .bind(ACCOUNT) + .bind(name) + .fetch_optional(&s.catalog) + .await + .expect("read the table status") +} + +#[tokio::test] +async fn legacy_composite_backup_restores_with_sort_keys() { + if base_conn().is_none() { + eprintln!("SKIP legacy_composite_backup_restores_with_sort_keys: no PostgreSQL"); + return; + } + let s = scratch().await; + let rows: Vec = (0..12) + .map(|i| { + serde_json::json!({ + "pk": {"S": format!("p{}", i % 3)}, + "sk": {"N": format!("{}", i * 10 - 50)}, + "v": {"S": format!("value-{i}")}, + }) + }) + .collect(); + let arn = legacy_backup(&s, "legacy_src", &rows).await; + + s.engine + .restore_table_from_backup( + ACCOUNT, + "legacy_dst", + &arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore a pre-fix backup"); + assert_eq!( + table_status(&s, "legacy_dst").await.as_deref(), + Some("ACTIVE") + ); + + let desc = s + .engine + .describe_table( + ACCOUNT, + DescribeTableInput { + table_name: "legacy_dst".to_owned(), + }, + ) + .await + .expect("describe the restored table"); + assert_eq!(desc.item_count, 12); + assert!( + desc.table_size_bytes > 0, + "a restored table reports its size as soon as it is ACTIVE" + ); + let key_info = TableKeyInfo { + table_name: "legacy_dst".to_owned(), + account_id: ACCOUNT.to_owned(), + table_id: desc.table_id.clone(), + key_schema: desc.key_schema.clone(), + base_key_schema: desc.key_schema.clone(), + attribute_definitions: desc.attribute_definitions.clone(), + ..Default::default() + }; + + // Every row is addressable by its full key, so the sort key landed in its + // typed column rather than being dropped. + for row in &rows { + let item: Item = serde_json::from_value(row.clone()).expect("item from json"); + let key: Item = item + .iter() + .filter(|(k, _)| *k == "pk" || *k == "sk") + .map(|(k, v)| (k.clone(), v.clone())) + .collect(); + let got = s + .engine + .get_item(&key_info, &key) + .await + .expect("point read"); + assert_eq!(got.as_ref(), Some(&item)); + } + + s.cleanup().await; +} + +#[tokio::test] +async fn failed_restore_removes_the_partial_table() { + if base_conn().is_none() { + eprintln!("SKIP failed_restore_removes_the_partial_table: no PostgreSQL"); + return; + } + let s = scratch().await; + let rows = vec![ + serde_json::json!({"pk": {"S": "a"}, "sk": {"N": "1"}}), + // No sort key: the copy cannot place this item. + serde_json::json!({"pk": {"S": "b"}}), + ]; + let arn = legacy_backup(&s, "broken_src", &rows).await; + // A non-zero control-plane delay sends an ordinary DeleteTable through + // DELETING; cleanup of a failed restore must not depend on that. + sqlx::query("UPDATE settings SET value = '5' WHERE key = 'control_plane_delay_seconds'") + .execute(&s.catalog) + .await + .expect("set a non-zero control-plane delay"); + + let err = s + .engine + .restore_table_from_backup( + ACCOUNT, + "broken_dst", + &arn, + RestoreTableOverrides::default(), + ) + .await + .expect_err("a backup row without its sort key cannot restore"); + assert!(matches!(err, StorageError::Internal(_)), "{err:?}"); + + // Neither CREATING, DELETING, nor half-filled: the target and its data + // table are gone and the name is free. + assert_eq!(table_status(&s, "broken_dst").await, None); + let leftover: i64 = sqlx::query_scalar( + "SELECT count(*) FROM pg_tables WHERE schemaname = 'public' AND tablename LIKE '\\_ddb\\_%'", + ) + .fetch_one(&s.db) + .await + .expect("count data tables"); + assert_eq!(leftover, 1, "only the source table's data table remains"); + s.engine + .create_table(ACCOUNT, composite_table("broken_dst")) + .await + .expect("the target name is free again"); + + s.cleanup().await; +} + +#[tokio::test] +async fn duplicate_backup_rows_fail_the_restore() { + if base_conn().is_none() { + eprintln!("SKIP duplicate_backup_rows_fail_the_restore: no PostgreSQL"); + return; + } + let s = scratch().await; + // 1 and 1.0 are one N key value: a backup carrying both is corrupt, and + // the restore must say so rather than keep one and report two. + let rows = vec![ + serde_json::json!({"pk": {"S": "a"}, "sk": {"N": "1"}, "v": {"S": "first"}}), + serde_json::json!({"pk": {"S": "a"}, "sk": {"N": "1.0"}, "v": {"S": "second"}}), + ]; + let arn = legacy_backup(&s, "dup_src", &rows).await; + let err = s + .engine + .restore_table_from_backup(ACCOUNT, "dup_dst", &arn, RestoreTableOverrides::default()) + .await + .expect_err("two rows for one key cannot restore"); + assert!(matches!(err, StorageError::Internal(_)), "{err:?}"); + assert_eq!(table_status(&s, "dup_dst").await, None); + s.cleanup().await; +} + +#[tokio::test] +async fn multipart_key_backup_is_refused() { + if base_conn().is_none() { + eprintln!("SKIP multipart_key_backup_is_refused: no PostgreSQL"); + return; + } + let s = scratch().await; + let key = |name: &str, key_type| KeySchemaElement { + attribute_name: name.to_owned(), + key_type, + }; + let attr = |name: &str, attribute_type| AttributeDefinition { + attribute_name: name.to_owned(), + attribute_type, + }; + let input = CreateTableInput { + table_name: "multi_src".to_owned(), + key_schema: vec![ + key("h1", KeyType::Hash), + key("h2", KeyType::Hash), + key("r1", KeyType::Range), + ], + attribute_definitions: vec![ + attr("h1", ScalarAttributeType::S), + attr("h2", ScalarAttributeType::N), + attr("r1", ScalarAttributeType::S), + ], + billing_mode: Some(BillingMode::PayPerRequest), + ..Default::default() + }; + let rows = vec![serde_json::json!({"h1": {"S": "a"}, "h2": {"N": "1"}, "r1": {"S": "x"}})]; + let arn = legacy_backup_of(&s, input, &rows).await; + // Multi-part base keys are a preview the item paths address only by their + // first parts, so no restore of one can be right; it must refuse, before + // creating anything. + let err = s + .engine + .restore_table_from_backup(ACCOUNT, "multi_dst", &arn, RestoreTableOverrides::default()) + .await + .expect_err("a multi-part key backup is refused"); + assert!(matches!(err, StorageError::Unsupported(_)), "{err:?}"); + assert_eq!(table_status(&s, "multi_dst").await, None); + s.cleanup().await; +} + +/// A provisioned (pk S, sk N) table with one GSI and one LSI, holding 30 +/// items, some of them outside each index. +async fn indexed_source(s: &Scratch, name: &str) -> String { + let input: CreateTableInput = serde_json::from_value(serde_json::json!({ + "TableName": name, + "KeySchema": [ + {"AttributeName": "pk", "KeyType": "HASH"}, + {"AttributeName": "sk", "KeyType": "RANGE"} + ], + "AttributeDefinitions": [ + {"AttributeName": "pk", "AttributeType": "S"}, + {"AttributeName": "sk", "AttributeType": "N"}, + {"AttributeName": "g", "AttributeType": "S"}, + {"AttributeName": "l", "AttributeType": "S"} + ], + "BillingMode": "PROVISIONED", + "ProvisionedThroughput": {"ReadCapacityUnits": 7, "WriteCapacityUnits": 9}, + "GlobalSecondaryIndexes": [{ + "IndexName": "gi", + "KeySchema": [{"AttributeName": "g", "KeyType": "HASH"}], + "Projection": {"ProjectionType": "KEYS_ONLY"}, + "ProvisionedThroughput": {"ReadCapacityUnits": 3, "WriteCapacityUnits": 4} + }], + "LocalSecondaryIndexes": [{ + "IndexName": "li", + "KeySchema": [ + {"AttributeName": "pk", "KeyType": "HASH"}, + {"AttributeName": "l", "KeyType": "RANGE"} + ], + "Projection": {"ProjectionType": "ALL"} + }] + })) + .expect("input"); + let desc = s + .engine + .create_table(ACCOUNT, input) + .await + .expect("create the source table"); + let key_info = TableKeyInfo { + table_name: name.to_owned(), + account_id: ACCOUNT.to_owned(), + table_id: desc.table_id.clone(), + key_schema: desc.key_schema.clone(), + base_key_schema: desc.key_schema.clone(), + attribute_definitions: desc.attribute_definitions.clone(), + ..Default::default() + }; + for i in 0..30 { + let mut item = + serde_json::json!({"pk": {"S": format!("p{}", i % 3)}, "sk": {"N": i.to_string()}}); + if i % 2 == 0 { + item["g"] = serde_json::json!({"S": format!("g{}", i % 4)}); + } + if i % 3 == 0 { + item["l"] = serde_json::json!({"S": format!("l{i}")}); + } + let item: Item = serde_json::from_value(item).expect("item"); + s.engine + .put_item( + &key_info, + item, + false, + None, + &extenddb_core::expression::ExpressionMaps::default(), + None, + ) + .await + .expect("put"); + } + desc.table_id +} + +/// `(index_name, index_id)` of a table's secondary indexes, by name. +async fn index_ids(s: &Scratch, table_id: &str) -> Vec<(String, String)> { + sqlx::query_as( + "SELECT index_name, index_id FROM indexes WHERE table_id = $1 ORDER BY index_name", + ) + .bind(table_id) + .fetch_all(&s.catalog) + .await + .expect("read indexes") +} + +async fn data_rows(s: &Scratch, physical_id: &str) -> i64 { + sqlx::query_scalar(&format!("SELECT count(*) FROM \"_ddb_{physical_id}\"")) + .fetch_one(&s.db) + .await + .expect("count rows") +} + +async fn data_table_exists(s: &Scratch, physical_id: &str) -> bool { + sqlx::query_scalar("SELECT to_regclass($1) IS NOT NULL") + .bind(format!("public.\"_ddb_{physical_id}\"")) + .fetch_one(&s.db) + .await + .expect("regclass") +} + +async fn table_id_of(s: &Scratch, name: &str) -> Option { + sqlx::query_scalar("SELECT table_id FROM tables WHERE account_id = $1 AND table_name = $2") + .bind(ACCOUNT) + .bind(name) + .fetch_optional(&s.catalog) + .await + .expect("table id") +} + +/// Create a one-row source and return its id and quoted physical table name. +async fn snapshot_source(s: &Scratch, name: &str) -> (String, String) { + let input = CreateTableInput { + table_name: name.to_owned(), + key_schema: vec![KeySchemaElement { + attribute_name: "pk".to_owned(), + key_type: KeyType::Hash, + }], + attribute_definitions: vec![AttributeDefinition { + attribute_name: "pk".to_owned(), + attribute_type: ScalarAttributeType::S, + }], + billing_mode: Some(BillingMode::PayPerRequest), + ..Default::default() + }; + let desc = s + .engine + .create_table(ACCOUNT, input) + .await + .expect("create snapshot source"); + let table = format!("\"_ddb_{}\"", desc.table_id); + insert_snapshot_item(&s.db, &table, "before").await; + (desc.table_id, table) +} + +async fn insert_snapshot_item(db: &PgPool, table: &str, key: &str) { + sqlx::query(&format!( + "INSERT INTO {table} (pk, item_data) VALUES ($1, $2)" + )) + .bind(key) + .bind(serde_json::json!({"pk": {"S": key}})) + .execute(db) + .await + .expect("insert snapshot item"); +} + +async fn snapshot_row_count(db: &PgPool, table: &str) -> i64 { + sqlx::query_scalar(&format!("SELECT count(*) FROM {table}")) + .fetch_one(db) + .await + .expect("count snapshot rows") +} + +#[tokio::test] +async fn restore_recreates_indexes_and_throughput() { + if base_conn().is_none() { + eprintln!("SKIP restore_recreates_indexes_and_throughput: no PostgreSQL"); + return; + } + let s = scratch().await; + let src = indexed_source(&s, "idx_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "idx_src", "b") + .await + .expect("backup"); + let desc = s + .engine + .restore_table_from_backup( + ACCOUNT, + "idx_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + assert_eq!(table_status(&s, "idx_dst").await.as_deref(), Some("ACTIVE")); + + let restored = s + .engine + .describe_table( + ACCOUNT, + DescribeTableInput { + table_name: "idx_dst".to_owned(), + }, + ) + .await + .expect("describe"); + let pt = restored.provisioned_throughput; + assert_eq!((pt.read_capacity_units, pt.write_capacity_units), (7, 9)); + let gsis = restored.global_secondary_indexes.expect("gsis"); + assert_eq!(gsis.len(), 1); + let gpt = gsis[0] + .provisioned_throughput + .as_ref() + .expect("gsi throughput"); + assert_eq!((gpt.read_capacity_units, gpt.write_capacity_units), (3, 4)); + assert_eq!(restored.local_secondary_indexes.map(|l| l.len()), Some(1)); + + // Each index holds exactly the rows the source's does. + let src_idx = index_ids(&s, &src).await; + let dst_idx = index_ids(&s, &desc.table_id).await; + assert_eq!( + dst_idx.iter().map(|(n, _)| n.as_str()).collect::>(), + ["gi", "li"] + ); + for ((_, a), (_, b)) in src_idx.iter().zip(&dst_idx) { + let want = data_rows(&s, a).await; + assert!(want > 0); + assert_eq!(data_rows(&s, b).await, want); + } + assert_eq!(data_rows(&s, &desc.table_id).await, 30); + s.cleanup().await; +} + +#[tokio::test] +async fn abandoned_restore_is_removed_only_when_unowned_and_old() { + if base_conn().is_none() { + eprintln!("SKIP abandoned_restore_is_removed_only_when_unowned_and_old: no PostgreSQL"); + return; + } + let s = scratch().await; + indexed_source(&s, "ab_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "ab_src", "b") + .await + .expect("backup"); + let desc = s + .engine + .restore_table_from_backup( + ACCOUNT, + "ab_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + let indexes = index_ids(&s, &desc.table_id).await; + + // The state a process killed mid-copy leaves: the target CREATING with no + // scheduled transition, its copy transaction rolled back. + let crash = |age: &'static str| { + let catalog = s.catalog.clone(); + let id = desc.table_id.clone(); + async move { + sqlx::query(&format!( + "UPDATE tables SET table_status = 'CREATING', status_transition_at = NULL, \ + creation_date_time = NOW() - INTERVAL '{age}' WHERE table_id = $1" + )) + .bind(&id) + .execute(&catalog) + .await + .expect("simulate a crash"); + } + }; + + // Inside the grace period: a restore that has not taken its lock yet. + crash("1 second").await; + s.engine + .process_control_plane_transitions() + .await + .expect("sweep"); + assert_eq!( + table_status(&s, "ab_dst").await.as_deref(), + Some("CREATING") + ); + + // Old, but its lock is held: a live restore on another instance. The key + // is the account and name, so the lock covers the target before it exists. + crash("10 minutes").await; + let mut owner = s.catalog.acquire().await.expect("connection"); + sqlx::query("SELECT pg_advisory_lock(hashtextextended($2, $1))") + .bind(0x0045_4452_i64) + .bind(format!("{ACCOUNT}/ab_dst")) + .execute(&mut *owner) + .await + .expect("hold the restore lock"); + s.engine + .process_control_plane_transitions() + .await + .expect("sweep"); + assert_eq!( + table_status(&s, "ab_dst").await.as_deref(), + Some("CREATING") + ); + sqlx::query("SELECT pg_advisory_unlock(hashtextextended($2, $1))") + .bind(0x0045_4452_i64) + .bind(format!("{ACCOUNT}/ab_dst")) + .execute(&mut *owner) + .await + .expect("release the restore lock"); + drop(owner); + + // Old and unowned: removed, with its data and index tables. + let transitions = s + .engine + .process_control_plane_transitions() + .await + .expect("sweep"); + assert!( + transitions.iter().any(|(n, _)| n == "ab_dst"), + "{transitions:?}" + ); + assert_eq!(table_status(&s, "ab_dst").await, None); + assert!(!data_table_exists(&s, &desc.table_id).await); + for (_, id) in &indexes { + assert!( + !data_table_exists(&s, id).await, + "index table {id} left behind" + ); + } + + // An ACTIVE table is never a candidate, and the name is free again. + s.engine + .restore_table_from_backup( + ACCOUNT, + "ab_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore again"); + sqlx::query("UPDATE tables SET creation_date_time = NOW() - INTERVAL '10 minutes'") + .execute(&s.catalog) + .await + .expect("age every table"); + s.engine + .process_control_plane_transitions() + .await + .expect("sweep"); + assert_eq!(table_status(&s, "ab_dst").await.as_deref(), Some("ACTIVE")); + s.cleanup().await; +} + +#[tokio::test] +async fn failed_restore_with_indexes_leaves_no_tables() { + if base_conn().is_none() { + eprintln!("SKIP failed_restore_with_indexes_leaves_no_tables: no PostgreSQL"); + return; + } + let s = scratch().await; + indexed_source(&s, "fi_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "fi_src", "b") + .await + .expect("backup"); + sqlx::query("INSERT INTO backup_items (backup_arn, pk, item_data) VALUES ($1, 'x', $2)") + .bind(&backup.backup_arn) + .bind(serde_json::json!({"pk": {"S": "x"}})) + .execute(&s.catalog) + .await + .expect("add a row without its sort key"); + let before: i64 = + sqlx::query_scalar("SELECT count(*) FROM pg_tables WHERE schemaname = 'public'") + .fetch_one(&s.db) + .await + .expect("tables"); + s.engine + .restore_table_from_backup( + ACCOUNT, + "fi_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect_err("a row without its sort key cannot restore"); + assert_eq!(table_id_of(&s, "fi_dst").await, None); + let after: i64 = + sqlx::query_scalar("SELECT count(*) FROM pg_tables WHERE schemaname = 'public'") + .fetch_one(&s.db) + .await + .expect("tables"); + assert_eq!( + after, before, + "the target's data and index tables are dropped" + ); + s.cleanup().await; +} + +#[tokio::test] +async fn restore_keeps_table_class_sse_and_on_demand_limits() { + if base_conn().is_none() { + eprintln!("SKIP restore_keeps_table_class_sse_and_on_demand_limits: no PostgreSQL"); + return; + } + let s = scratch().await; + let input: CreateTableInput = serde_json::from_value(serde_json::json!({ + "TableName": "cls_src", + "KeySchema": [{"AttributeName": "pk", "KeyType": "HASH"}], + "AttributeDefinitions": [{"AttributeName": "pk", "AttributeType": "S"}], + "BillingMode": "PAY_PER_REQUEST", + "TableClass": "STANDARD_INFREQUENT_ACCESS", + "SSESpecification": {"Enabled": true, "SSEType": "KMS"}, + "OnDemandThroughput": {"MaxReadRequestUnits": 100, "MaxWriteRequestUnits": 50} + })) + .expect("input"); + let src = s.engine.create_table(ACCOUNT, input).await.expect("create"); + let backup = s + .engine + .create_backup(ACCOUNT, "cls_src", "b") + .await + .expect("backup"); + s.engine + .restore_table_from_backup( + ACCOUNT, + "cls_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + let read = |name: &'static str| { + let catalog = s.catalog.clone(); + async move { + sqlx::query_as::< + _, + ( + String, + Option, + Option, + Option, + ), + >( + "SELECT billing_mode, table_class, sse_specification, on_demand_throughput \ + FROM tables WHERE account_id = $1 AND table_name = $2", + ) + .bind(ACCOUNT) + .bind(name) + .fetch_one(&catalog) + .await + .expect("row") + } + }; + let want = read("cls_src").await; + assert_eq!(want.1.as_deref(), Some("STANDARD_INFREQUENT_ACCESS")); + assert!(want.2.is_some() && want.3.is_some(), "{want:?}"); + assert_eq!(read("cls_dst").await, want); + drop(src); + s.cleanup().await; +} + +#[tokio::test] +async fn restore_crosses_batch_boundaries() { + if base_conn().is_none() { + eprintln!("SKIP restore_crosses_batch_boundaries: no PostgreSQL"); + return; + } + let s = scratch().await; + let src = indexed_source(&s, "many_src").await; + // More rows than one item batch, and a few large enough that the byte + // budget, not the row count, ends some batches. + let desc_rows: Vec = (0..1_234) + .map(|i| { + let mut v = serde_json::json!({ + "pk": {"S": format!("bulk{}", i % 11)}, + "sk": {"N": format!("{}", 1000 + i)}, + "g": {"S": format!("g{}", i % 4)}, + }); + if i % 97 == 0 { + v["pad"] = serde_json::json!({"S": "x".repeat(300_000)}); + } + v + }) + .collect(); + let backup = s + .engine + .create_backup(ACCOUNT, "many_src", "b") + .await + .expect("backup"); + sqlx::query( + "INSERT INTO backup_items (backup_arn, pk, item_data) \ + SELECT $1, '', d FROM UNNEST($2::jsonb[]) AS t(d)", + ) + .bind(&backup.backup_arn) + .bind(&desc_rows) + .execute(&s.catalog) + .await + .expect("add rows to the backup"); + let desc = s + .engine + .restore_table_from_backup( + ACCOUNT, + "many_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + assert_eq!(data_rows(&s, &desc.table_id).await, 30 + 1_234); + let src_gsi = &index_ids(&s, &src).await[0].1; + let dst_gsi = &index_ids(&s, &desc.table_id).await[0].1; + assert_eq!( + data_rows(&s, dst_gsi).await, + data_rows(&s, src_gsi).await + 1_234 + ); + s.cleanup().await; +} + +#[tokio::test] +async fn restore_into_a_name_whose_restore_lock_is_held_is_refused() { + if base_conn().is_none() { + eprintln!("SKIP restore_into_a_name_whose_restore_lock_is_held_is_refused: no PostgreSQL"); + return; + } + let s = scratch().await; + indexed_source(&s, "lk_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "lk_src", "b") + .await + .expect("backup"); + let mut other = s.catalog.acquire().await.expect("connection"); + sqlx::query("SELECT pg_advisory_lock(hashtextextended($2, $1))") + .bind(0x0045_4452_i64) + .bind(format!("{ACCOUNT}/lk_dst")) + .execute(&mut *other) + .await + .expect("another restore owns the name"); + let err = s + .engine + .restore_table_from_backup( + ACCOUNT, + "lk_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect_err("the name is being restored into already"); + assert!( + matches!(err, StorageError::TableAlreadyExists(_)), + "{err:?}" + ); + assert_eq!(table_status(&s, "lk_dst").await, None); + sqlx::query("SELECT pg_advisory_unlock_all()") + .execute(&mut *other) + .await + .expect("release"); + drop(other); + s.cleanup().await; +} + +/// A backup taken before catalog 0.0.4 has no definition row. Rolling the +/// catalog back to that shape, replaying migration 003 twice (as an upgrade +/// interrupted between applying and recording it would), and restoring must +/// give the table as restores did before: keys and items, no secondary +/// indexes, and 5/5 throughput for a provisioned table. +#[tokio::test] +async fn pre_0_0_4_backup_restores_after_migrating() { + if base_conn().is_none() { + eprintln!("SKIP pre_0_0_4_backup_restores_after_migrating: no PostgreSQL"); + return; + } + let s = scratch().await; + indexed_source(&s, "old_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "old_src", "b") + .await + .expect("backup"); + sqlx::query("DROP TABLE backup_definitions") + .execute(&s.catalog) + .await + .expect("roll the catalog back to 0.0.3"); + for _ in 0..2 { + sqlx::raw_sql(include_str!("../migrations/003_backup_definitions.sql")) + .execute(&s.catalog) + .await + .expect("apply migration 003"); + } + let version: String = + sqlx::query_scalar("SELECT value FROM settings WHERE key = 'catalog_version'") + .fetch_one(&s.catalog) + .await + .expect("version"); + assert_eq!(version, "0.0.4"); + + let desc = s + .engine + .restore_table_from_backup( + ACCOUNT, + "old_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore a pre-0.0.4 backup"); + assert_eq!(table_status(&s, "old_dst").await.as_deref(), Some("ACTIVE")); + assert!(index_ids(&s, &desc.table_id).await.is_empty()); + assert_eq!(data_rows(&s, &desc.table_id).await, 30); + let restored = s + .engine + .describe_table( + ACCOUNT, + DescribeTableInput { + table_name: "old_dst".to_owned(), + }, + ) + .await + .expect("describe"); + assert_eq!( + ( + restored.provisioned_throughput.read_capacity_units, + restored.provisioned_throughput.write_capacity_units + ), + (5, 5) + ); + s.cleanup().await; +} + +/// A real CreateBackup pins its item snapshot before its final catalog INSERT. +/// Holding that INSERT proves exactly when the snapshot must already exist: +/// an item committed while the INSERT waits is live, but absent from the backup. +#[tokio::test] +async fn production_create_backup_excludes_item_committed_after_snapshot_pin() { + if base_conn().is_none() { + eprintln!( + "SKIP production_create_backup_excludes_item_committed_after_snapshot_pin: no PostgreSQL" + ); + return; + } + let s = scratch().await; + let (_, table) = snapshot_source(&s, "prod_snap").await; + + let mut gate = s.catalog.begin().await.expect("begin backups gate"); + sqlx::query("LOCK TABLE backups IN ACCESS EXCLUSIVE MODE") + .execute(&mut *gate) + .await + .expect("block CreateBackup's catalog insert"); + + let mut backup_fut = s.engine.create_backup(ACCOUNT, "prod_snap", "b"); + let early = tokio::time::timeout(std::time::Duration::from_secs(2), &mut backup_fut).await; + assert!( + early.is_err(), + "CreateBackup did not wait at its final backups INSERT" + ); + let insert_is_waiting: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM pg_locks \ + WHERE relation = 'backups'::regclass AND NOT granted)", + ) + .fetch_one(&s.catalog) + .await + .expect("observe the blocked backup insert"); + assert!( + insert_is_waiting, + "CreateBackup was blocked somewhere other than its final backups INSERT" + ); + + // This commit is after the production path's relation SELECT, but before + // its first item cursor. Without that SELECT, the later cursor is the first + // snapshot-establishing statement and incorrectly includes this row. + insert_snapshot_item(&s.db, &table, "after").await; + gate.commit().await.expect("release the backup insert"); + let backup = backup_fut.await.expect("finish the production backup"); + + let backed_up: i64 = + sqlx::query_scalar("SELECT count(*) FROM backup_items WHERE backup_arn = $1") + .bind(&backup.backup_arn) + .fetch_one(&s.catalog) + .await + .expect("count production backup items"); + assert_eq!(backed_up, 1, "the backup included a post-snapshot item"); + assert_eq!( + snapshot_row_count(&s.db, &table).await, + 2, + "the concurrent item must have committed" + ); + s.cleanup().await; +} + +/// DeleteBackup wins the backup-row lock before restore registers its target. +/// Restore must wait, then observe the deleted backup and leave no target. +#[tokio::test] +async fn delete_backup_winning_before_restore_leaves_no_target() { + if base_conn().is_none() { + eprintln!("SKIP delete_backup_winning_before_restore_leaves_no_target: no PostgreSQL"); + return; + } + let s = scratch().await; + snapshot_source(&s, "delete_first_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "delete_first_src", "b") + .await + .expect("backup"); + + // DeleteBackup locks the backup FOR UPDATE before reading table_restores. + // Gate that read so the lock is held while restore reaches its FOR SHARE. + let mut gate = s.catalog.begin().await.expect("begin provenance gate"); + sqlx::query("LOCK TABLE table_restores IN ACCESS EXCLUSIVE MODE") + .execute(&mut *gate) + .await + .expect("block DeleteBackup after its row lock"); + let mut delete_fut = s.engine.delete_backup(ACCOUNT, &backup.backup_arn); + let early_delete = + tokio::time::timeout(std::time::Duration::from_secs(2), &mut delete_fut).await; + assert!(early_delete.is_err(), "DeleteBackup did not reach the gate"); + let delete_is_waiting: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM pg_locks \ + WHERE relation = 'table_restores'::regclass AND NOT granted)", + ) + .fetch_one(&s.catalog) + .await + .expect("observe blocked DeleteBackup"); + assert!( + delete_is_waiting, + "DeleteBackup was not paused after its row lock" + ); + + let mut restore_fut = s.engine.restore_table_from_backup( + ACCOUNT, + "delete_first_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ); + let early_restore = + tokio::time::timeout(std::time::Duration::from_millis(500), &mut restore_fut).await; + assert!( + early_restore.is_err(), + "restore did not wait for DeleteBackup's row lock" + ); + assert_eq!( + table_id_of(&s, "delete_first_dst").await, + None, + "restore created its target before locking the backup" + ); + + gate.commit().await.expect("let DeleteBackup finish"); + delete_fut.await.expect("DeleteBackup wins the ordering"); + let err = restore_fut + .await + .expect_err("restore must observe that the backup was deleted"); + assert!( + matches!(&err, StorageError::Validation(message) if message.contains("Backup not found")), + "{err:?}" + ); + assert_eq!(table_id_of(&s, "delete_first_dst").await, None); + s.cleanup().await; +} + +/// Restore commits its CREATING target and provenance before copying items. +/// DeleteBackup must then see the two rows together and return BackupInUse. +#[tokio::test] +async fn restore_registration_winning_before_delete_returns_backup_in_use() { + if base_conn().is_none() { + eprintln!( + "SKIP restore_registration_winning_before_delete_returns_backup_in_use: no PostgreSQL" + ); + return; + } + let s = scratch().await; + snapshot_source(&s, "restore_first_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "restore_first_src", "b") + .await + .expect("backup"); + + let mut gate = s.catalog.begin().await.expect("begin item-copy gate"); + sqlx::query("LOCK TABLE backup_items IN ACCESS EXCLUSIVE MODE") + .execute(&mut *gate) + .await + .expect("pause restore after target registration"); + let mut restore_fut = s.engine.restore_table_from_backup( + ACCOUNT, + "restore_first_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ); + let early_restore = + tokio::time::timeout(std::time::Duration::from_secs(2), &mut restore_fut).await; + assert!( + early_restore.is_err(), + "restore did not reach the item-copy gate" + ); + let copy_is_waiting: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM pg_locks \ + WHERE relation = 'backup_items'::regclass AND NOT granted)", + ) + .fetch_one(&s.catalog) + .await + .expect("observe blocked restore copy"); + assert!(copy_is_waiting, "restore was not paused in its item copy"); + + // One statement observes both rows, after the create-table transaction has + // committed but before the copy can finish. + let registered: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM tables t \ + JOIN table_restores r ON r.table_id = t.table_id \ + WHERE t.account_id = $1 AND t.table_name = $2 \ + AND t.table_status = 'CREATING' AND r.source_backup_arn = $3)", + ) + .bind(ACCOUNT) + .bind("restore_first_dst") + .bind(&backup.backup_arn) + .fetch_one(&s.catalog) + .await + .expect("observe atomic restore registration"); + assert!(registered, "target and provenance did not commit together"); + + let err = s + .engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect_err("a registered in-progress restore keeps its backup in use"); + assert!(matches!(err, StorageError::BackupInUse(_)), "{err:?}"); + + gate.commit().await.expect("release the restore copy"); + let desc = restore_fut + .await + .expect("restore completes after the gate opens"); + assert_eq!( + table_status(&s, "restore_first_dst").await.as_deref(), + Some("ACTIVE") + ); + assert_eq!(data_rows(&s, &desc.table_id).await, 1); + s.cleanup().await; +} + +/// LOCK TABLE does not establish a REPEATABLE READ snapshot. This is the old +/// CreateBackup shape: after the catalog barrier would have been released, a +/// committed insert is visible to the first item SELECT. +#[tokio::test] +async fn backup_snapshot_with_lock_only_includes_later_insert() { + if base_conn().is_none() { + eprintln!("SKIP backup_snapshot_with_lock_only_includes_later_insert: no PostgreSQL"); + return; + } + let s = scratch().await; + let (_, table) = snapshot_source(&s, "old_snap").await; + assert_eq!(snapshot_row_count(&s.db, &table).await, 1); + + let mut snapshot = + s.db.begin_with("BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY") + .await + .expect("begin old-shape snapshot"); + sqlx::query(&format!("LOCK TABLE {table} IN ACCESS SHARE MODE")) + .execute(&mut *snapshot) + .await + .expect("lock the source relation"); + + // This commits after the point where CreateBackup released its catalog + // barrier. Because no SELECT fixed the snapshot, the item read sees it. + insert_snapshot_item(&s.db, &table, "after").await; + let visible: i64 = sqlx::query_scalar(&format!("SELECT count(*) FROM {table}")) + .fetch_one(&mut *snapshot) + .await + .expect("read through the old snapshot shape"); + assert_eq!(visible, 2, "LOCK TABLE unexpectedly fixed the snapshot"); + snapshot.commit().await.expect("commit snapshot"); + s.cleanup().await; +} + +/// A real relation SELECT fixes the backup's REPEATABLE READ snapshot before +/// the catalog barrier is released. A later committed item is live in the +/// database but absent from every subsequent read through that snapshot. +#[tokio::test] +async fn backup_snapshot_with_relation_read_excludes_later_insert() { + if base_conn().is_none() { + eprintln!("SKIP backup_snapshot_with_relation_read_excludes_later_insert: no PostgreSQL"); + return; + } + let s = scratch().await; + let (_, table) = snapshot_source(&s, "new_snap").await; + + let mut snapshot = + s.db.begin_with("BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY") + .await + .expect("begin backup-shaped snapshot"); + sqlx::query(&format!("LOCK TABLE {table} IN ACCESS SHARE MODE")) + .execute(&mut *snapshot) + .await + .expect("lock the source relation"); + let first: Option = sqlx::query_scalar(&format!("SELECT 1 FROM {table} LIMIT 1")) + .fetch_optional(&mut *snapshot) + .await + .expect("fix the snapshot by reading the relation"); + assert_eq!(first, Some(1)); + + // This is the first write after the catalog barrier is released. + insert_snapshot_item(&s.db, &table, "after").await; + let visible: i64 = sqlx::query_scalar(&format!("SELECT count(*) FROM {table}")) + .fetch_one(&mut *snapshot) + .await + .expect("read through the fixed snapshot"); + assert_eq!(visible, 1, "the fixed snapshot included a later item"); + snapshot.commit().await.expect("commit snapshot"); + + // Liveness beside the isolation assertion: the writer did commit and the + // current database state moved, even though the backup snapshot did not. + assert_eq!(snapshot_row_count(&s.db, &table).await, 2); + s.cleanup().await; +} + +/// A backup's ACCESS SHARE lock can delay DROP TABLE, but it must not make the +/// control-plane pass wait indefinitely or lose the DELETING row needed to +/// retry once the backup releases its snapshot. +#[tokio::test] +async fn control_plane_drop_timeout_leaves_table_deleting_for_retry() { + if base_conn().is_none() { + eprintln!("SKIP control_plane_drop_timeout_leaves_table_deleting_for_retry: no PostgreSQL"); + return; + } + let s = scratch().await; + let (table_id, table) = snapshot_source(&s, "drop_retry").await; + sqlx::query( + "UPDATE tables SET table_status = 'DELETING', status_transition_at = NOW() \ + WHERE table_id = $1", + ) + .bind(&table_id) + .execute(&s.catalog) + .await + .expect("schedule table deletion"); + + let mut backup = + s.db.begin_with("BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY") + .await + .expect("begin backup-shaped snapshot"); + sqlx::query(&format!("LOCK TABLE {table} IN ACCESS SHARE MODE")) + .execute(&mut *backup) + .await + .expect("hold the backup table lock"); + let _: Option = sqlx::query_scalar(&format!("SELECT 1 FROM {table} LIMIT 1")) + .fetch_optional(&mut *backup) + .await + .expect("fix the backup snapshot"); + + let first = tokio::time::timeout( + std::time::Duration::from_secs(15), + s.engine.process_control_plane_transitions(), + ) + .await + .expect("blocked DROP must respect its lock timeout") + .expect("a timed-out drop is retriable, not a failed worker pass"); + assert!(!first.iter().any(|(name, _)| name == "drop_retry")); + assert_eq!( + table_status(&s, "drop_retry").await.as_deref(), + Some("DELETING") + ); + assert!(data_table_exists(&s, &table_id).await); + + backup.commit().await.expect("release the backup lock"); + let second = s + .engine + .process_control_plane_transitions() + .await + .expect("retry the data drop"); + assert!( + second.iter().any(|(name, _)| name == "drop_retry"), + "{second:?}" + ); + assert_eq!(table_status(&s, "drop_retry").await, None); + assert!(!data_table_exists(&s, &table_id).await); + s.cleanup().await; +} + +/// The table row is held FOR SHARE until the data snapshot is taken, so a +/// definition change cannot commit between the two: an UpdateTable blocked +/// on the row waits for the backup's barrier, and the backup records the +/// definition in force when its items were read. +#[tokio::test] +async fn backup_definition_and_items_share_one_instant() { + if base_conn().is_none() { + eprintln!("SKIP backup_definition_and_items_share_one_instant: no PostgreSQL"); + return; + } + let s = scratch().await; + indexed_source(&s, "bar_src").await; + let table_id = table_id_of(&s, "bar_src").await.expect("table"); + + // Hold the row the way UpdateTable does, and change the billing mode + // without committing yet. + let mut writer = s.catalog.begin().await.expect("writer"); + sqlx::query("SELECT 1 FROM tables WHERE table_id = $1 FOR UPDATE") + .bind(&table_id) + .execute(&mut *writer) + .await + .expect("lock the row"); + sqlx::query("UPDATE tables SET billing_mode = 'PAY_PER_REQUEST' WHERE table_id = $1") + .bind(&table_id) + .execute(&mut *writer) + .await + .expect("change the definition"); + + // The backup must wait for the writer rather than read around it. + let arn = { + let backup = s.engine.create_backup(ACCOUNT, "bar_src", "b"); + tokio::pin!(backup); + let early = tokio::time::timeout(std::time::Duration::from_millis(500), &mut backup).await; + assert!( + early.is_err(), + "CreateBackup read the definition past an uncommitted change" + ); + writer.commit().await.expect("commit the change"); + backup.await.expect("backup").backup_arn + }; + + let def: serde_json::Value = + sqlx::query_scalar("SELECT definition FROM backup_definitions WHERE backup_arn = $1") + .bind(&arn) + .fetch_one(&s.catalog) + .await + .expect("definition"); + assert_eq!(def["BillingMode"], "PAY_PER_REQUEST"); + s.cleanup().await; +} + +/// UpdateTable commits a new GSI's catalog row before it builds the index's +/// data table. A backup taken in between leaves that index out (it is not +/// part of the table yet), drops the attribute definition only it used, and +/// restores. +#[tokio::test] +async fn backup_leaves_out_a_gsi_still_being_built() { + if base_conn().is_none() { + eprintln!("SKIP backup_leaves_out_a_gsi_still_being_built: no PostgreSQL"); + return; + } + let s = scratch().await; + let src = indexed_source(&s, "unb_src").await; + let gi = index_ids(&s, &src) + .await + .into_iter() + .find(|(n, _)| n == "gi") + .expect("gi") + .1; + sqlx::query(&format!("DROP TABLE \"_ddb_{gi}\"")) + .execute(&s.db) + .await + .expect("make the GSI look half-built"); + let backup = s + .engine + .create_backup(ACCOUNT, "unb_src", "b") + .await + .expect("backup"); + let def: serde_json::Value = + sqlx::query_scalar("SELECT definition FROM backup_definitions WHERE backup_arn = $1") + .bind(&backup.backup_arn) + .fetch_one(&s.catalog) + .await + .expect("definition"); + assert_eq!(def["GlobalSecondaryIndexes"], serde_json::json!([])); + let attrs: serde_json::Value = + sqlx::query_scalar("SELECT attribute_definitions FROM backups WHERE backup_arn = $1") + .bind(&backup.backup_arn) + .fetch_one(&s.catalog) + .await + .expect("attrs"); + let names: Vec<&str> = attrs + .as_array() + .expect("array") + .iter() + .filter_map(|a| a["AttributeName"].as_str()) + .collect(); + assert!(!names.contains(&"g"), "{names:?}"); + assert!(names.contains(&"l"), "the LSI still uses l: {names:?}"); + + let desc = s + .engine + .restore_table_from_backup( + ACCOUNT, + "unb_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + assert_eq!(table_status(&s, "unb_dst").await.as_deref(), Some("ACTIVE")); + let names: Vec = index_ids(&s, &desc.table_id) + .await + .into_iter() + .map(|(n, _)| n) + .collect(); + assert_eq!(names, ["li"]); + s.cleanup().await; +} + +#[tokio::test] +async fn delete_table_refuses_a_restore_in_progress() { + if base_conn().is_none() { + eprintln!("SKIP delete_table_refuses_a_restore_in_progress: no PostgreSQL"); + return; + } + let s = scratch().await; + indexed_source(&s, "dt_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "dt_src", "b") + .await + .expect("backup"); + let desc = s + .engine + .restore_table_from_backup( + ACCOUNT, + "dt_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + sqlx::query( + "UPDATE tables SET table_status = 'CREATING', status_transition_at = NULL \ + WHERE table_id = $1", + ) + .bind(&desc.table_id) + .execute(&s.catalog) + .await + .expect("back to in progress"); + let err = s + .engine + .delete_table( + ACCOUNT, + extenddb_core::types::DeleteTableInput { + table_name: "dt_dst".to_owned(), + }, + ) + .await + .expect_err("refused"); + assert!(matches!(err, StorageError::IndexesInUse(_)), "{err:?}"); + assert_eq!( + table_status(&s, "dt_dst").await.as_deref(), + Some("CREATING") + ); + s.cleanup().await; +} + +/// DescribeTable reports where a restored table came from, in progress +/// while it is CREATING and done once ACTIVE; DeleteBackup is refused while a +/// restore from the backup is still running, and allowed afterwards, with +/// the summary still naming the deleted backup. +#[tokio::test] +async fn restore_summary_and_backup_in_use() { + if base_conn().is_none() { + eprintln!("SKIP restore_summary_and_backup_in_use: no PostgreSQL"); + return; + } + let s = scratch().await; + indexed_source(&s, "rs_src").await; + let backup = s + .engine + .create_backup(ACCOUNT, "rs_src", "b") + .await + .expect("backup"); + let desc = s + .engine + .restore_table_from_backup( + ACCOUNT, + "rs_dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + let describe = |name: &'static str| { + let engine = &s.engine; + async move { + engine + .describe_table( + ACCOUNT, + DescribeTableInput { + table_name: name.to_owned(), + }, + ) + .await + .expect("describe") + } + }; + let done = describe("rs_dst").await.restore_summary.expect("summary"); + assert_eq!( + done.source_backup_arn.as_deref(), + Some(backup.backup_arn.as_str()) + ); + assert!(!done.restore_in_progress); + assert!(done.restore_date_time > 0.0); + assert!(describe("rs_src").await.restore_summary.is_none()); + + // While the restore is still running: in progress, and the backup in use. + sqlx::query( + "UPDATE tables SET table_status = 'CREATING', status_transition_at = NULL \ + WHERE table_id = $1", + ) + .bind(&desc.table_id) + .execute(&s.catalog) + .await + .expect("back to in progress"); + assert!( + describe("rs_dst") + .await + .restore_summary + .expect("summary") + .restore_in_progress + ); + let err = s + .engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect_err("in use"); + assert!(matches!(err, StorageError::BackupInUse(_)), "{err:?}"); + + sqlx::query("UPDATE tables SET table_status = 'ACTIVE' WHERE table_id = $1") + .bind(&desc.table_id) + .execute(&s.catalog) + .await + .expect("finished"); + s.engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect("deletable once the restore is done"); + let after = describe("rs_dst").await.restore_summary.expect("summary"); + assert_eq!( + after.source_backup_arn.as_deref(), + Some(backup.backup_arn.as_str()) + ); + s.cleanup().await; +} diff --git a/crates/storage-postgres/tests/key_collation.rs b/crates/storage-postgres/tests/key_collation.rs index 263972dc9..9d1ba4a58 100644 --- a/crates/storage-postgres/tests/key_collation.rs +++ b/crates/storage-postgres/tests/key_collation.rs @@ -93,6 +93,7 @@ async fn scratch() -> Scratch { for sql in [ include_str!("../migrations/001_schema.sql"), include_str!("../migrations/002_vector_indexes.sql"), + include_str!("../migrations/003_backup_definitions.sql"), include_str!("../data_migrations/001_data_schema.sql"), include_str!("../data_migrations/002_gsi_pending.sql"), include_str!("../data_migrations/003_idempotency_account_scope.sql"), diff --git a/crates/storage-postgres/tests/vector_control_plane.rs b/crates/storage-postgres/tests/vector_control_plane.rs index 336011702..cfa4fad14 100644 --- a/crates/storage-postgres/tests/vector_control_plane.rs +++ b/crates/storage-postgres/tests/vector_control_plane.rs @@ -32,7 +32,7 @@ use extenddb_core::types::{ VectorIndexUpdate, }; use extenddb_storage::error::StorageError; -use extenddb_storage::{BackupEngine, DataEngine, TableEngine}; +use extenddb_storage::{BackupEngine, DataEngine, RestoreTableOverrides, TableEngine}; use extenddb_storage_postgres::{PostgresConfig, PostgresEngine}; use sqlx::PgPool; use sqlx::postgres::PgPoolOptions; @@ -160,6 +160,7 @@ async fn scratch(pgvector: Pgvector) -> Scratch { for sql in [ include_str!("../migrations/001_schema.sql"), include_str!("../migrations/002_vector_indexes.sql"), + include_str!("../migrations/003_backup_definitions.sql"), include_str!("../data_migrations/001_data_schema.sql"), include_str!("../data_migrations/002_gsi_pending.sql"), include_str!("../data_migrations/003_idempotency_account_scope.sql"), @@ -917,7 +918,12 @@ async fn restoring_a_backup_that_carries_vector_indexes_is_refused() { // and whose client only finds out on the first search. let err = s .engine - .restore_table_from_backup(ACCOUNT, "t_restored", &details.backup_arn) + .restore_table_from_backup( + ACCOUNT, + "t_restored", + &details.backup_arn, + RestoreTableOverrides::default(), + ) .await .expect_err("restoring a vector-indexed backup must be refused"); match err { @@ -950,7 +956,12 @@ async fn restoring_a_backup_that_carries_vector_indexes_is_refused() { .await .expect("back up a plain table"); s.engine - .restore_table_from_backup(ACCOUNT, "t_plain_restored", &plain.backup_arn) + .restore_table_from_backup( + ACCOUNT, + "t_plain_restored", + &plain.backup_arn, + RestoreTableOverrides::default(), + ) .await .expect("a backup with no vector indexes must still restore"); @@ -1270,7 +1281,12 @@ async fn a_backup_taken_before_the_snapshot_column_existed_still_restores() { .expect("blank the snapshot the way a pre-migration backup has it"); s.engine - .restore_table_from_backup(ACCOUNT, "t_legacy_restored", &details.backup_arn) + .restore_table_from_backup( + ACCOUNT, + "t_legacy_restored", + &details.backup_arn, + RestoreTableOverrides::default(), + ) .await .expect("a pre-migration backup must still restore"); diff --git a/crates/storage-sqlite/src/backup.rs b/crates/storage-sqlite/src/backup.rs index b79422a21..0f54bef47 100644 --- a/crates/storage-sqlite/src/backup.rs +++ b/crates/storage-sqlite/src/backup.rs @@ -3,23 +3,33 @@ //! `BackupEngine` for the SQLite backend. //! -//! A backup snapshots every item's `item_data` into `backup_items`. Restore -//! recreates the table via `create_table` and upserts the snapshot under the -//! engine write lock. `RestoreTableToPointInTime` is implemented as a +//! A backup snapshots every item's `item_data` into `backup_items` and the +//! table's definition (secondary indexes, billing mode and throughput, table +//! class, encryption) into `backup_definitions`. Restore recreates the table +//! with that definition and copies the snapshot, filling the secondary +//! indexes as it goes, under the engine write lock. `RestoreTableToPointInTime` is implemented as a //! snapshot-then-restore (then discard the temporary backup), matching the //! PostgreSQL backend's behaviour. use extenddb_core::types::{ AttributeDefinition, BackupDescription, BackupDetails, BackupSummary, BillingMode, - ContinuousBackupsDescription, CreateTableInput, KeySchemaElement, - PointInTimeRecoveryDescription, ProvisionedThroughput, SourceTableDetails, TableDescription, - TableKeyInfo, + ContinuousBackupsDescription, CreateTableInput, GsiInput, Item, KeySchemaElement, LsiInput, + PointInTimeRecoveryDescription, Projection, ProvisionedThroughput, SourceTableDetails, + TableDescription, +}; +use extenddb_storage::backup_definition::{ + BACKUP_DEFINITION_VERSION, BackupTableDefinition, COPY_BATCH_BYTES, COPY_BATCH_ITEMS, + ensure_single_part_base_key, throughput_from_catalog, }; -use extenddb_storage::BackupEngine; use extenddb_storage::error::StorageError; +use extenddb_storage::util::{composite_pk_to_text, parse_sk, sk_column_n}; +use extenddb_storage::{BackupEngine, RestoreTableOverrides}; use futures::future::BoxFuture; -use crate::data::{data_table_name, upsert_item_in_tx}; +use crate::data::{ + BoundValue, all_sort_key_info, data_table_name, index_table_name, insert_index_row_multi, + item_has_index_keys, project_item_for_index, sk_bound, +}; use crate::sqlite_util::parse_timestamp; use crate::store::SqliteEngine; @@ -46,100 +56,789 @@ fn ts_to_epoch(s: &str) -> f64 { .unwrap_or(0.0) } -impl BackupEngine for SqliteEngine { - fn create_backup( +/// Read the next batch of `(cursor, item_data)` from `table` after +/// `last_cursor`, restricted by `filter` (a fixed SQL condition with one `?` +/// bound to `filter_arg`, or `None`). `cursor` is `rowid` for immutable source +/// table snapshots and `id` for backup_items, whose values survive VACUUM. +/// +/// Two statements: the first reads only row lengths, so the batch can be cut +/// to the byte budget before any item text is loaded. Both run in the caller's +/// transaction, so no row appears or moves between them. +async fn next_batch( + tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>, + table: &str, + cursor: &str, + filter: Option<(&str, &str)>, + last_cursor: i64, +) -> Result, StorageError> { + let (cond, arg) = filter.map_or(("1 = 1", None), |(c, a)| (c, Some(a))); + let sizes_sql = format!( + "SELECT {cursor}, length(CAST(item_data AS BLOB)) FROM {table} \ + WHERE {cond} AND {cursor} > ? ORDER BY {cursor} LIMIT ?" + ); + let mut q = sqlx::query_as::<_, (i64, i64)>(&sizes_sql); + if let Some(a) = arg { + q = q.bind(a); + } + let sizes = q + .bind(last_cursor) + .bind(i64::try_from(COPY_BATCH_ITEMS).unwrap_or(i64::MAX)) + .fetch_all(&mut **tx) + .await + .map_err(db_err)?; + let Some(&(first, _)) = sizes.first() else { + return Ok(Vec::new()); + }; + let mut upto = first; + let mut bytes: usize = 0; + for &(value, len) in &sizes { + let len = usize::try_from(len).unwrap_or(usize::MAX); + if value != first && bytes.saturating_add(len) > COPY_BATCH_BYTES { + break; + } + bytes = bytes.saturating_add(len); + upto = value; + } + let rows_sql = format!( + "SELECT {cursor}, item_data FROM {table} \ + WHERE {cond} AND {cursor} > ? AND {cursor} <= ? ORDER BY {cursor}" + ); + let mut q = sqlx::query_as::<_, (i64, String)>(&rows_sql); + if let Some(a) = arg { + q = q.bind(a); + } + q.bind(last_cursor) + .bind(upto) + .fetch_all(&mut **tx) + .await + .map_err(db_err) +} + +fn db_err(e: sqlx::Error) -> StorageError { + StorageError::Internal(e.to_string()) +} + +fn parse_json(s: &str, what: &str) -> Result { + serde_json::from_str(s).map_err(|e| StorageError::Internal(format!("{what}: {e}"))) +} + +/// A secondary index of a restore target, as the copy needs it. +struct RestoreIndex { + table: String, + key_schema: Vec, + projection: Projection, +} + +impl SqliteEngine { + /// Read the parts of a table's definition a restore recreates. + async fn capture_table_definition( + tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>, + table_id: &str, + ) -> Result { + let (billing_mode, pt, table_class, sse, on_demand): ( + String, + Option, + Option, + Option, + Option, + ) = sqlx::query_as( + "SELECT billing_mode, provisioned_throughput, table_class, sse_specification, \ + on_demand_throughput FROM tables WHERE table_id = ?", + ) + .bind(table_id) + .fetch_one(&mut **tx) + .await + .map_err(db_err)?; + + let rows: Vec<(String, String, String, String, Option)> = sqlx::query_as( + "SELECT index_name, index_type, key_schema, projection, provisioned_throughput \ + FROM indexes WHERE table_id = ? ORDER BY index_name", + ) + .bind(table_id) + .fetch_all(&mut **tx) + .await + .map_err(db_err)?; + let mut gsis = Vec::new(); + let mut lsis = Vec::new(); + for (index_name, index_type, ks, proj, ipt) in rows { + let key_schema: Vec = parse_json(&ks, "index key schema")?; + let projection: Projection = parse_json(&proj, "index projection")?; + if index_type == "LSI" { + lsis.push(LsiInput { + index_name, + key_schema, + projection, + }); + } else { + let ipt: Option = ipt + .as_deref() + .map(|s| parse_json(s, "index throughput")) + .transpose()?; + gsis.push(GsiInput { + index_name, + key_schema, + projection, + provisioned_throughput: throughput_from_catalog(ipt.as_ref()), + }); + } + } + + let vector_index_names: Vec = sqlx::query_scalar( + "SELECT index_name FROM vector_indexes WHERE table_id = ? ORDER BY index_name", + ) + .bind(table_id) + .fetch_all(&mut **tx) + .await + .map_err(db_err)?; + + let pt: Option = pt + .as_deref() + .map(|s| parse_json(s, "throughput")) + .transpose()?; + Ok(BackupTableDefinition { + version: BACKUP_DEFINITION_VERSION, + billing_mode, + provisioned_throughput: throughput_from_catalog(pt.as_ref()), + global_secondary_indexes: gsis, + local_secondary_indexes: lsis, + table_class, + sse_specification: sse.as_deref().map(|s| parse_json(s, "sse")).transpose()?, + on_demand_throughput: on_demand + .as_deref() + .map(|s| parse_json(s, "on-demand throughput")) + .transpose()?, + vector_index_names, + }) + } + + /// Copy a backup's items into a freshly created restore target and fill + /// its secondary indexes, inside the caller's write transaction. + /// + /// Every key column is derived from the item (all HASH parts into `pk`, + /// each RANGE part into its typed column), and each row is a plain + /// INSERT, so two backup rows for one key fail the restore instead of + /// collapsing into one item. Items are read and written in byte-bounded + /// batches, each in its own write transaction, and the target is flipped + /// to ACTIVE in the last one. + async fn copy_backup_items( + &self, + desc: &TableDescription, + backup_arn: &str, + ) -> Result<(), StorageError> { + let ddb_table = data_table_name(&desc.table_id); + let sort_keys = all_sort_key_info(&desc.key_schema, &desc.attribute_definitions); + let mut cols = vec!["pk".to_owned()]; + cols.extend( + sort_keys + .iter() + .enumerate() + .map(|(i, &(_, t))| sk_column_n(i, t)), + ); + cols.push("item_data".to_owned()); + let insert_sql = format!( + "INSERT INTO {ddb_table} ({}) VALUES ({})", + cols.join(", "), + vec!["?"; cols.len()].join(", ") + ); + + let index_rows: Vec<(String, String, String)> = sqlx::query_as( + "SELECT index_id, key_schema, projection FROM indexes WHERE table_id = ?", + ) + .bind(&desc.table_id) + .fetch_all(&self.pool) + .await + .map_err(db_err)?; + let mut indexes = Vec::with_capacity(index_rows.len()); + for (index_id, ks, proj) in index_rows { + indexes.push(RestoreIndex { + table: index_table_name(&index_id), + key_schema: parse_json(&ks, "index key schema")?, + projection: parse_json(&proj, "index projection")?, + }); + } + let base_sks = all_sort_key_info(&desc.key_schema, &desc.attribute_definitions); + + let mut last_rowid: i64 = 0; + let mut count: i64 = 0; + loop { + // One write transaction per batch, so other writers on the server + // wait for a batch rather than the whole restore. The target is + // CREATING throughout, which refuses every data-plane request, so + // no one sees it part-filled. Each batch first re-checks, under + // the write lock, that the backup is still AVAILABLE (DeleteBackup + // removes its rows and marks it DELETED in one transaction) and + // that the target is still this restore's. + let _writer = self.write_lock.lock().await; + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") + .await + .map_err(db_err)?; + Self::ensure_restore_can_continue(&mut tx, desc, backup_arn).await?; + let batch = next_batch( + &mut tx, + "backup_items", + "id", + Some(("backup_arn = ?", backup_arn)), + last_rowid, + ) + .await?; + let Some(&(tail, _)) = batch.last() else { + let (table_size,): (i64,) = sqlx::query_as(&format!( + "SELECT COALESCE(SUM(length(item_data)), 0) FROM {ddb_table}" + )) + .fetch_one(&mut *tx) + .await + .map_err(db_err)?; + // Only from CREATING, which the check above confirmed under + // the same lock. + sqlx::query( + "UPDATE tables SET item_count = ?, table_size_bytes = ?, \ + table_status = 'ACTIVE', status_transition_at = NULL \ + WHERE table_id = ? AND table_status = 'CREATING'", + ) + .bind(count) + .bind(table_size) + .bind(&desc.table_id) + .execute(&mut *tx) + .await + .map_err(db_err)?; + tx.commit().await.map_err(db_err)?; + return Ok(()); + }; + last_rowid = tail; + for (_, item_json) in &batch { + let item: Item = parse_json(item_json, "backup item")?; + let mut values = vec![BoundValue::Text(composite_pk_to_text( + &item, + &desc.key_schema, + )?)]; + for &(name, sk_type) in &sort_keys { + let value = item.get(name).ok_or_else(|| { + StorageError::Internal(format!( + "backup {backup_arn} has an item without sort key {name}" + )) + })?; + values.push(sk_bound(&parse_sk(value, sk_type)?)); + } + values.push(BoundValue::Text(item_json.clone())); + let mut q = sqlx::query(&insert_sql); + for v in values { + q = crate::data::bind_bound!(q, v); + } + q.execute(&mut *tx).await.map_err(db_err)?; + + for idx in &indexes { + if !item_has_index_keys(&item, &idx.key_schema) { + continue; + } + let idx_sks = all_sort_key_info(&idx.key_schema, &desc.attribute_definitions); + let projected = project_item_for_index( + &item, + &idx.key_schema, + &desc.key_schema, + &idx.projection, + ); + insert_index_row_multi( + &mut tx, + &idx.table, + &item, + &projected, + &idx.key_schema, + &desc.key_schema, + &idx_sks, + &base_sks, + ) + .await?; + } + count += 1; + } + tx.commit().await.map_err(db_err)?; + } + } + + /// Fail a restore whose backup was deleted or whose target is no longer + /// the one it created. + async fn ensure_restore_can_continue( + tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>, + desc: &TableDescription, + backup_arn: &str, + ) -> Result<(), StorageError> { + let available: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM backups WHERE backup_arn = ? \ + AND backup_status = 'AVAILABLE')", + ) + .bind(backup_arn) + .fetch_one(&mut **tx) + .await + .map_err(db_err)?; + if !available { + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + } + let still_target: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM tables t \ + JOIN table_restores r ON r.table_id = t.table_id \ + WHERE t.table_id = ? AND r.source_backup_arn = ? \ + AND t.table_status = 'CREATING' AND t.status_transition_at IS NULL)", + ) + .bind(&desc.table_id) + .bind(backup_arn) + .fetch_one(&mut **tx) + .await + .map_err(db_err)?; + if !still_target { + return Err(StorageError::TableNotFound(format!( + "restore of {backup_arn} into {} did not complete: the table was deleted \ + while it was being restored", + desc.table_name + ))); + } + Ok(()) + } + + /// Remove a restore target whose copy failed or was abandoned: its + /// catalog row (cascading its index rows) and every data table. Only a + /// target still CREATING with no scheduled transition is removed, and by + /// table id, so this can never touch a table the restore did not create + /// or one a client already deleted. + async fn abort_restore(&self, table_id: &str) -> Result { + let _writer = self.write_lock.lock().await; + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") + .await + .map_err(db_err)?; + let index_ids: Vec = + sqlx::query_scalar("SELECT index_id FROM indexes WHERE table_id = ?") + .bind(table_id) + .fetch_all(&mut *tx) + .await + .map_err(db_err)?; + let removed = sqlx::query( + "DELETE FROM tables WHERE table_id = ? AND table_status = 'CREATING' \ + AND status_transition_at IS NULL", + ) + .bind(table_id) + .execute(&mut *tx) + .await + .map_err(db_err)? + .rows_affected() + > 0; + if removed { + for id in &index_ids { + Self::drop_index_data_table(&mut tx, id).await?; + } + Self::drop_data_table(&mut tx, table_id).await?; + } + tx.commit().await.map_err(db_err)?; + Ok(removed) + } + + /// Create a file-backed backup from one WAL snapshot while writers run + /// between bounded backup_items insert batches. + async fn create_backup_from_wal_snapshot( &self, account_id: &str, table_name: &str, backup_name: &str, - ) -> BoxFuture<'_, Result> { - let account_id = account_id.to_owned(); - let table_name = table_name.to_owned(); - let backup_name = backup_name.to_owned(); - Box::pin(async move { - let row: Option<(String, String, String, String, i64)> = sqlx::query_as( - "SELECT table_id, key_schema, attribute_definitions, billing_mode, table_size_bytes \ - FROM tables WHERE account_id = ? AND table_name = ? AND table_status = 'ACTIVE'", - ) - .bind(&account_id) - .bind(&table_name) - .fetch_optional(&self.pool) + backup_arn: &str, + ) -> Result { + // The writer lock is held only while the definition is read and while + // the item snapshot is established. No engine writer can change the + // table between those two instants. Later backup_items inserts take the + // lock for one bounded batch each, so unrelated writers do not wait for + // the whole backup. + let writer = self.write_lock.lock().await; + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") .await + .map_err(db_err)?; + let row: Option<(String, String, String, String)> = sqlx::query_as( + "SELECT table_id, key_schema, attribute_definitions, billing_mode \ + FROM tables WHERE account_id = ? AND table_name = ? AND table_status = 'ACTIVE'", + ) + .bind(account_id) + .bind(table_name) + .fetch_optional(&mut *tx) + .await + .map_err(db_err)?; + let (table_id, key_schema_json, attr_defs, billing_mode) = + row.ok_or_else(|| StorageError::TableNotFound(table_name.to_owned()))?; + let key_schema: Vec = parse_json(&key_schema_json, "key schema")?; + let definition = Self::capture_table_definition(&mut tx, &table_id).await?; + let definition = serde_json::to_string(&definition.to_json()?) .map_err(|e| StorageError::Internal(e.to_string()))?; - let (table_id, key_schema, attr_defs, billing_mode, size_bytes) = - row.ok_or_else(|| StorageError::TableNotFound(table_name.clone()))?; + sqlx::query( + "INSERT INTO backups (backup_arn, backup_name, table_id, table_name, account_id, \ + backup_status, backup_size_bytes, item_count, key_schema, attribute_definitions, \ + billing_mode) VALUES (?, ?, ?, ?, ?, 'CREATING', 0, 0, ?, ?, ?)", + ) + .bind(backup_arn) + .bind(backup_name) + .bind(&table_id) + .bind(table_name) + .bind(account_id) + .bind(&key_schema_json) + .bind(&attr_defs) + .bind(&billing_mode) + .execute(&mut *tx) + .await + .map_err(db_err)?; + sqlx::query("INSERT INTO backup_definitions (backup_arn, definition) VALUES (?, ?)") + .bind(backup_arn) + .bind(&definition) + .execute(&mut *tx) + .await + .map_err(db_err)?; + let created_at: String = + sqlx::query_scalar("SELECT created_at FROM backups WHERE backup_arn = ?") + .bind(backup_arn) + .fetch_one(&mut *tx) + .await + .map_err(db_err)?; + tx.commit().await.map_err(db_err)?; - let backup_arn = format!( - "arn:aws:dynamodb:{region}:{account_id}:table/{table_name}/backup/{id}", - region = self.region, - id = backup_id() - ); + let ddb_table = data_table_name(&table_id); + let mut read_tx = self.pool.begin().await.map_err(db_err)?; + // BEGIN is deferred. This first table read fixes its WAL snapshot while + // the writer lock still excludes every engine write, so the definition, + // source row and items describe one table state. + let establish_snapshot = format!("SELECT 1 FROM {ddb_table} LIMIT 1"); + let _: Option = sqlx::query_scalar(&establish_snapshot) + .fetch_optional(&mut *read_tx) + .await + .map_err(db_err)?; + drop(writer); - let ddb_table = data_table_name(&table_id); - let items: Vec<(String,)> = - sqlx::query_as(&format!("SELECT item_data FROM {ddb_table}")) - .fetch_all(&self.pool) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - let item_count = i64::try_from(items.len()).unwrap_or(i64::MAX); + let mut last_rowid = 0_i64; + let mut item_count = 0_i64; + let mut size_bytes = 0_i64; + loop { + let batch = next_batch(&mut read_tx, &ddb_table, "rowid", None, last_rowid).await?; + let Some(&(tail, _)) = batch.last() else { + break; + }; + last_rowid = tail; + + let mut prepared = Vec::with_capacity(batch.len()); + for (_, item_data) in batch { + let item: Item = parse_json(&item_data, "item")?; + let pk = composite_pk_to_text(&item, &key_schema)?; + let size = + i64::try_from(extenddb_core::types::item_size_bytes(&item)).unwrap_or(i64::MAX); + prepared.push((pk, item_data, size)); + } let _writer = self.write_lock.lock().await; - let mut tx = self + let mut write_tx = self .pool .begin_with("BEGIN IMMEDIATE") .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - - sqlx::query( - "INSERT INTO backups (backup_arn, backup_name, table_id, table_name, account_id, \ - backup_status, backup_size_bytes, item_count, key_schema, attribute_definitions, \ - billing_mode) VALUES (?, ?, ?, ?, ?, 'AVAILABLE', ?, ?, ?, ?, ?)", + .map_err(db_err)?; + let creating: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM backups \ + WHERE backup_arn = ? AND backup_status = 'CREATING')", ) - .bind(&backup_arn) - .bind(&backup_name) - .bind(&table_id) - .bind(&table_name) - .bind(&account_id) - .bind(size_bytes) - .bind(item_count) - .bind(&key_schema) - .bind(&attr_defs) - .bind(&billing_mode) - .execute(&mut *tx) + .bind(backup_arn) + .fetch_one(&mut *write_tx) + .await + .map_err(db_err)?; + if !creating { + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + } + for (pk, item_data, size) in prepared { + sqlx::query( + "INSERT INTO backup_items (backup_arn, pk, sk, item_data) \ + VALUES (?, ?, NULL, ?)", + ) + .bind(backup_arn) + .bind(pk) + .bind(item_data) + .execute(&mut *write_tx) + .await + .map_err(db_err)?; + item_count = item_count.saturating_add(1); + size_bytes = size_bytes.saturating_add(size); + } + write_tx.commit().await.map_err(db_err)?; + } + read_tx.commit().await.map_err(db_err)?; + + let _writer = self.write_lock.lock().await; + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") + .await + .map_err(db_err)?; + let updated = sqlx::query( + "UPDATE backups SET backup_status = 'AVAILABLE', item_count = ?, \ + backup_size_bytes = ? WHERE backup_arn = ? AND backup_status = 'CREATING'", + ) + .bind(item_count) + .bind(size_bytes) + .bind(backup_arn) + .execute(&mut *tx) + .await + .map_err(db_err)? + .rows_affected(); + if updated != 1 { + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + } + tx.commit().await.map_err(db_err)?; + + Ok(BackupDetails { + backup_arn: backup_arn.to_owned(), + backup_name: backup_name.to_owned(), + backup_status: "AVAILABLE".to_owned(), + backup_type: "USER".to_owned(), + backup_size_bytes: size_bytes, + backup_creation_date_time: ts_to_epoch(&created_at), + }) + } + + /// In-memory SQLite has one connection, so it cannot hold a read snapshot + /// and open a writer transaction concurrently. Keep its backup atomic on + /// that connection; there is no second connection whose writes could make + /// progress, and item memory remains bounded by one batch. + async fn create_backup_single_connection( + &self, + account_id: &str, + table_name: &str, + backup_name: &str, + backup_arn: &str, + ) -> Result { + let _writer = self.write_lock.lock().await; + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") .await + .map_err(db_err)?; + let row: Option<(String, String, String, String)> = sqlx::query_as( + "SELECT table_id, key_schema, attribute_definitions, billing_mode \ + FROM tables WHERE account_id = ? AND table_name = ? AND table_status = 'ACTIVE'", + ) + .bind(account_id) + .bind(table_name) + .fetch_optional(&mut *tx) + .await + .map_err(db_err)?; + let (table_id, key_schema_json, attr_defs, billing_mode) = + row.ok_or_else(|| StorageError::TableNotFound(table_name.to_owned()))?; + let key_schema: Vec = parse_json(&key_schema_json, "key schema")?; + let definition = Self::capture_table_definition(&mut tx, &table_id).await?; + let definition = serde_json::to_string(&definition.to_json()?) .map_err(|e| StorageError::Internal(e.to_string()))?; + sqlx::query( + "INSERT INTO backups (backup_arn, backup_name, table_id, table_name, account_id, \ + backup_status, backup_size_bytes, item_count, key_schema, attribute_definitions, \ + billing_mode) VALUES (?, ?, ?, ?, ?, 'CREATING', 0, 0, ?, ?, ?)", + ) + .bind(backup_arn) + .bind(backup_name) + .bind(&table_id) + .bind(table_name) + .bind(account_id) + .bind(&key_schema_json) + .bind(&attr_defs) + .bind(&billing_mode) + .execute(&mut *tx) + .await + .map_err(db_err)?; + sqlx::query("INSERT INTO backup_definitions (backup_arn, definition) VALUES (?, ?)") + .bind(backup_arn) + .bind(definition) + .execute(&mut *tx) + .await + .map_err(db_err)?; - for (item_data,) in &items { + let ddb_table = data_table_name(&table_id); + let mut last_rowid = 0_i64; + let mut item_count = 0_i64; + let mut size_bytes = 0_i64; + loop { + let batch = next_batch(&mut tx, &ddb_table, "rowid", None, last_rowid).await?; + let Some(&(tail, _)) = batch.last() else { + break; + }; + last_rowid = tail; + for (_, item_data) in batch { + let item: Item = parse_json(&item_data, "item")?; + let pk = composite_pk_to_text(&item, &key_schema)?; + size_bytes = size_bytes.saturating_add( + i64::try_from(extenddb_core::types::item_size_bytes(&item)).unwrap_or(i64::MAX), + ); sqlx::query( - "INSERT INTO backup_items (backup_arn, pk, sk, item_data) VALUES (?, '', NULL, ?)", + "INSERT INTO backup_items (backup_arn, pk, sk, item_data) \ + VALUES (?, ?, NULL, ?)", ) - .bind(&backup_arn) + .bind(backup_arn) + .bind(pk) .bind(item_data) .execute(&mut *tx) .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + .map_err(db_err)?; + item_count = item_count.saturating_add(1); } + } + sqlx::query( + "UPDATE backups SET backup_status = 'AVAILABLE', item_count = ?, \ + backup_size_bytes = ? WHERE backup_arn = ? AND backup_status = 'CREATING'", + ) + .bind(item_count) + .bind(size_bytes) + .bind(backup_arn) + .execute(&mut *tx) + .await + .map_err(db_err)?; + let created_at: String = + sqlx::query_scalar("SELECT created_at FROM backups WHERE backup_arn = ?") + .bind(backup_arn) + .fetch_one(&mut *tx) + .await + .map_err(db_err)?; + tx.commit().await.map_err(db_err)?; - let created_at: (String,) = - sqlx::query_as("SELECT created_at FROM backups WHERE backup_arn = ?") - .bind(&backup_arn) - .fetch_one(&mut *tx) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + Ok(BackupDetails { + backup_arn: backup_arn.to_owned(), + backup_name: backup_name.to_owned(), + backup_status: "AVAILABLE".to_owned(), + backup_type: "USER".to_owned(), + backup_size_bytes: size_bytes, + backup_creation_date_time: ts_to_epoch(&created_at), + }) + } - tx.commit() + async fn remove_incomplete_backup(&self, backup_arn: &str) -> Result { + let _writer = self.write_lock.lock().await; + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") + .await + .map_err(db_err)?; + let removed = + sqlx::query("DELETE FROM backups WHERE backup_arn = ? AND backup_status = 'CREATING'") + .bind(backup_arn) + .execute(&mut *tx) .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + .map_err(db_err)? + .rows_affected() + > 0; + tx.commit().await.map_err(db_err)?; + Ok(removed) + } - Ok(BackupDetails { - backup_arn, - backup_name, - backup_status: "AVAILABLE".to_owned(), - backup_type: "USER".to_owned(), - backup_size_bytes: size_bytes, - backup_creation_date_time: ts_to_epoch(&created_at.0), - }) + /// Remove backups left CREATING by a process that died during its copy. + /// The backups delete cascades to definitions and any copied item rows. + pub async fn sweep_incomplete_backups(&self) -> Result, StorageError> { + let _writer = self.write_lock.lock().await; + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") + .await + .map_err(db_err)?; + let removed: Vec = sqlx::query_scalar( + "DELETE FROM backups WHERE backup_status = 'CREATING' RETURNING backup_arn", + ) + .fetch_all(&mut *tx) + .await + .map_err(db_err)?; + tx.commit().await.map_err(db_err)?; + Ok(removed) + } + + /// Remove restore targets left by a process that died mid-restore. + /// + /// A restore target is the only table that sits CREATING with no + /// scheduled transition. Called once at startup, before any request is + /// served, so no restore of this process can be in flight and every such + /// table is abandoned. + /// + /// This assumes one server process per database file, which the backend + /// already depends on: writers are serialized by an in-process lock, so a + /// second process writing the same file is unsupported regardless. A + /// second process started against the file while the first is mid-restore + /// would remove that restore's target. + pub(crate) async fn sweep_abandoned_restores(&self) -> Result, StorageError> { + let candidates: Vec<(String, String)> = sqlx::query_as( + "SELECT table_id, table_name FROM tables \ + WHERE table_status = 'CREATING' AND status_transition_at IS NULL", + ) + .fetch_all(&self.pool) + .await + .map_err(db_err)?; + let mut removed = Vec::new(); + for (table_id, table_name) in candidates { + if self.abort_restore(&table_id).await? { + removed.push(table_name); + } + } + Ok(removed) + } +} + +impl BackupEngine for SqliteEngine { + fn create_backup( + &self, + account_id: &str, + table_name: &str, + backup_name: &str, + ) -> BoxFuture<'_, Result> { + let engine = self.clone(); + let account_id = account_id.to_owned(); + let table_name = table_name.to_owned(); + let backup_name = backup_name.to_owned(); + Box::pin(async move { + // Detach the operation from the request future. Once the CREATING + // row commits, dropping an HTTP request must not drop the copy and + // strand wire-visible state. A dropped JoinHandle leaves its Tokio + // task running, so the copy reaches AVAILABLE or executes cleanup. + let task = tokio::spawn(async move { + let backup_arn = format!( + "arn:aws:dynamodb:{region}:{account_id}:table/{table_name}/backup/{id}", + region = engine.region, + id = backup_id() + ); + let created = if engine.in_memory { + engine + .create_backup_single_connection( + &account_id, + &table_name, + &backup_name, + &backup_arn, + ) + .await + } else { + engine + .create_backup_from_wal_snapshot( + &account_id, + &table_name, + &backup_name, + &backup_arn, + ) + .await + }; + if let Err(error) = created { + if let Err(cleanup) = engine.remove_incomplete_backup(&backup_arn).await { + tracing::error!( + "backup {backup_arn} failed ({error}); could not remove its CREATING \ + row: {cleanup}; the next startup will retry cleanup" + ); + } + return Err(error); + } + created + }); + task.await.map_err(|error| { + StorageError::Internal(format!("SQLite backup task failed: {error}")) + })? }) } @@ -269,6 +968,11 @@ impl BackupEngine for SqliteEngine { // Resolves account-scoped, so a backup owned by another account is // reported missing here and the writes below never run. let desc = self.describe_backup(&account_id, &backup_arn).await?; + if desc.backup_details.backup_status != "AVAILABLE" { + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + } // D1: every writer holds the engine write lock. The item delete // below is a bulk statement (one row per backed-up item), which is @@ -276,16 +980,47 @@ impl BackupEngine for SqliteEngine { // past a concurrent locked writer's busy_timeout. let _writer = self.write_lock.lock().await; - // The account predicate is repeated on both writes rather than - // relying on the lookup above, so the statements are correct on - // their own terms. + // One transaction, so a restore never sees the backup AVAILABLE + // with its rows gone. The account predicate is repeated on every + // write rather than relying on the lookup above, so the + // statements are correct on their own terms. + let mut tx = self + .pool + .begin_with("BEGIN IMMEDIATE") + .await + .map_err(db_err)?; + // Refuse while a restore from this backup is still running, as the + // service does. Under the write lock, which the restore also takes + // to record itself, so the check cannot miss one. + let restoring: Option = sqlx::query_scalar( + "SELECT t.table_name FROM table_restores r JOIN tables t ON t.table_id = r.table_id \ + WHERE r.source_backup_arn = ? AND t.table_status = 'CREATING' LIMIT 1", + ) + .bind(&backup_arn) + .fetch_optional(&mut *tx) + .await + .map_err(db_err)?; + if let Some(table) = restoring { + return Err(StorageError::BackupInUse(format!( + "Backup is being used to restore table {table}: {backup_arn}" + ))); + } sqlx::query( "DELETE FROM backup_items WHERE backup_arn = ?1 AND EXISTS (\ SELECT 1 FROM backups b WHERE b.backup_arn = ?1 AND b.account_id = ?2)", ) .bind(&backup_arn) .bind(&account_id) - .execute(&self.pool) + .execute(&mut *tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + sqlx::query( + "DELETE FROM backup_definitions WHERE backup_arn = ?1 AND EXISTS (\ + SELECT 1 FROM backups b WHERE b.backup_arn = ?1 AND b.account_id = ?2)", + ) + .bind(&backup_arn) + .bind(&account_id) + .execute(&mut *tx) .await .map_err(|e| StorageError::Internal(e.to_string()))?; sqlx::query( @@ -294,9 +1029,10 @@ impl BackupEngine for SqliteEngine { ) .bind(&backup_arn) .bind(&account_id) - .execute(&self.pool) + .execute(&mut *tx) .await .map_err(|e| StorageError::Internal(e.to_string()))?; + tx.commit().await.map_err(db_err)?; Ok(BackupDescription { backup_details: BackupDetails { backup_status: "DELETED".to_owned(), @@ -312,122 +1048,102 @@ impl BackupEngine for SqliteEngine { account_id: &str, target_table_name: &str, backup_arn: &str, + overrides: RestoreTableOverrides, ) -> BoxFuture<'_, Result> { let account_id = account_id.to_owned(); let target_table_name = target_table_name.to_owned(); let backup_arn = backup_arn.to_owned(); Box::pin(async move { - let row: Option<(String, String, String)> = sqlx::query_as( - "SELECT key_schema, attribute_definitions, billing_mode \ - FROM backups WHERE backup_arn = ? AND account_id = ? \ - AND backup_status = 'AVAILABLE'", + let row: Option<(String, String, String, Option)> = sqlx::query_as( + "SELECT b.key_schema, b.attribute_definitions, b.billing_mode, d.definition \ + FROM backups b LEFT JOIN backup_definitions d ON d.backup_arn = b.backup_arn \ + WHERE b.backup_arn = ? AND b.account_id = ? AND b.backup_status = 'AVAILABLE'", ) .bind(&backup_arn) .bind(&account_id) .fetch_optional(&self.pool) .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + .map_err(db_err)?; - let (ks, ad, billing) = row.ok_or_else(|| { + let (ks, ad, billing, definition) = row.ok_or_else(|| { StorageError::Validation(format!("Backup not found: {backup_arn}")) })?; - let key_schema: Vec = - serde_json::from_str(&ks).map_err(|e| StorageError::Internal(e.to_string()))?; - let attr_defs: Vec = - serde_json::from_str(&ad).map_err(|e| StorageError::Internal(e.to_string()))?; - let billing_mode = Some(if billing == "PAY_PER_REQUEST" { - BillingMode::PayPerRequest - } else { - BillingMode::Provisioned - }); - - let create_input = CreateTableInput { - table_name: target_table_name.clone(), - key_schema: key_schema.clone(), - attribute_definitions: attr_defs.clone(), - billing_mode, - provisioned_throughput: Some(ProvisionedThroughput { - read_capacity_units: 5, - write_capacity_units: 5, - }), - global_secondary_indexes: None, - local_secondary_indexes: None, - stream_specification: None, - tags: None, - deletion_protection_enabled: None, - sse_specification: None, - table_class: None, - on_demand_throughput: None, - ..Default::default() - }; + let definition = definition + .map(|d| { + BackupTableDefinition::from_json( + parse_json(&d, "table definition")?, + &backup_arn, + ) + }) + .transpose()?; + if let Some(d) = &definition { + d.ensure_restorable(&backup_arn)?; + } + let key_schema: Vec = parse_json(&ks, "key schema")?; + ensure_single_part_base_key(&key_schema, &backup_arn)?; + let attr_defs: Vec = parse_json(&ad, "attribute definitions")?; - // `defer_active`: the target is written CREATING with no scheduled - // transition, so the control-plane worker cannot report it ACTIVE - // while the copy below is still running. The explicit ACTIVE update - // after the copy commits is the only flip. - let desc = self - .create_table_impl(&account_id, create_input, true) - .await?; - let key_info = TableKeyInfo { + let mut create_input = CreateTableInput { table_name: target_table_name.clone(), - account_id: account_id.clone(), - table_id: desc.table_id.clone(), - base_key_schema: key_schema.clone(), key_schema, attribute_definitions: attr_defs, - has_lsi: false, - // The restored table is created above without secondary - // indexes, so there is no index metadata to carry. - global_secondary_indexes: Vec::new(), - local_secondary_indexes: Vec::new(), - stream_specification: None, ..Default::default() }; + match definition { + Some(d) => d.apply_to(&mut create_input, &overrides)?, + None => { + // A backup taken before table definitions were recorded + // carries only its keys and billing mode, so it restores as + // it always has: no secondary indexes, and 5/5 throughput for + // a provisioned table since the source's is unknown. + let on_demand = billing == "PAY_PER_REQUEST"; + create_input.billing_mode = Some(if on_demand { + BillingMode::PayPerRequest + } else { + BillingMode::Provisioned + }); + create_input.provisioned_throughput = + (!on_demand).then_some(ProvisionedThroughput { + read_capacity_units: 5, + write_capacity_units: 5, + }); + overrides.apply_to_create_input(&mut create_input); + } + } - let items: Vec<(String,)> = - sqlx::query_as("SELECT item_data FROM backup_items WHERE backup_arn = ?") - .bind(&backup_arn) - .fetch_all(&self.pool) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - let item_count = i64::try_from(items.len()).unwrap_or(i64::MAX); + // The backup availability check, target row, and restore provenance + // commit under one write-lock hold. DeleteBackup therefore sees + // either no target or a CREATING target that names this backup. + let desc = self + .create_table_for_restore(&account_id, create_input, &backup_arn) + .await?; - let _writer = self.write_lock.lock().await; - let mut tx = self - .pool - .begin_with("BEGIN IMMEDIATE") - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - for (item_json,) in &items { - let item: extenddb_core::types::Item = serde_json::from_str(item_json) - .map_err(|e| StorageError::Internal(e.to_string()))?; - upsert_item_in_tx(&mut tx, &key_info, &item).await?; + // The copy commits in batches and flips the target ACTIVE in its + // last one; until then the target is CREATING and refuses every + // data-plane request. A failure part-way is cleaned up here, a + // crash part-way by the startup sweep. + let copied = self.copy_backup_items(&desc, &backup_arn).await; + if let Err(e) = copied { + tracing::error!( + "restore of {backup_arn} into {target_table_name} failed, \ + removing the partial table: {e}" + ); + if let Err(cleanup) = self.abort_restore(&desc.table_id).await { + tracing::error!( + "could not remove partially restored table {target_table_name} \ + ({}); the next startup will: {cleanup}", + desc.table_id + ); + } + return Err(e); } - sqlx::query("UPDATE tables SET item_count = ? WHERE account_id = ? AND table_name = ?") - .bind(item_count) - .bind(&account_id) - .bind(&target_table_name) - .execute(&mut *tx) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - // Mark the restored table ACTIVE immediately: the data is fully - // populated and ready to serve. This matches the Postgres backend - // and real DynamoDB, where a restored table becomes ACTIVE once the - // restore completes (the CREATING status is transient) rather than - // waiting for the control-plane transition poller. - sqlx::query( - "UPDATE tables SET table_status = 'ACTIVE', status_transition_at = NULL \ - WHERE account_id = ? AND table_name = ?", - ) - .bind(&account_id) - .bind(&target_table_name) - .execute(&mut *tx) - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; - tx.commit() - .await - .map_err(|e| StorageError::Internal(e.to_string()))?; + // The summary carries the recorded restore time, so the response + // and later DescribeTable calls agree. + let mut desc = desc; + desc.restore_summary = self + .restore_summary(&desc.table_id, &desc.table_status) + .await?; Ok(desc) }) } @@ -535,10 +1251,1276 @@ impl BackupEngine for SqliteEngine { .create_backup(&account_id, &source_table_name, "__pitr_restore__") .await?; let desc = self - .restore_table_from_backup(&account_id, &target_table_name, &backup.backup_arn) + .restore_table_from_backup( + &account_id, + &target_table_name, + &backup.backup_arn, + RestoreTableOverrides::default(), + ) .await?; let _ = self.delete_backup(&account_id, &backup.backup_arn).await; Ok(desc) }) } } + +#[cfg(test)] +mod restore_tests { + use extenddb_core::types::{Item, TableKeyInfo}; + use extenddb_storage::{BackupEngine, DataEngine, RestoreTableOverrides}; + use serde_json::json; + + use crate::SqliteEngine; + + const ACCOUNT: &str = "000000000000"; + + async fn engine() -> SqliteEngine { + let engine = SqliteEngine::new(":memory:", 1, "us-east-1", 409_600) + .await + .expect("engine"); + crate::schema::apply(&engine.pool).await.expect("schema"); + sqlx::query("UPDATE settings SET value = '0' WHERE key = 'control_plane_delay_seconds'") + .execute(&engine.pool) + .await + .expect("delay"); + sqlx::query("UPDATE settings SET value = '0' WHERE key = 'index_propagation_delay_ms'") + .execute(&engine.pool) + .await + .expect("propagation"); + sqlx::query("INSERT INTO accounts (account_id, account_name) VALUES (?, 'default')") + .bind(ACCOUNT) + .execute(&engine.pool) + .await + .expect("account"); + engine + } + + /// A provisioned two-part-key table with one GSI and one LSI. + async fn indexed_table(engine: &SqliteEngine, name: &str) -> String { + let input: extenddb_core::types::CreateTableInput = serde_json::from_value(json!({ + "TableName": name, + "KeySchema": [ + {"AttributeName": "pk", "KeyType": "HASH"}, + {"AttributeName": "sk", "KeyType": "RANGE"} + ], + "AttributeDefinitions": [ + {"AttributeName": "pk", "AttributeType": "S"}, + {"AttributeName": "sk", "AttributeType": "N"}, + {"AttributeName": "g", "AttributeType": "S"}, + {"AttributeName": "l", "AttributeType": "S"} + ], + "BillingMode": "PROVISIONED", + "ProvisionedThroughput": {"ReadCapacityUnits": 7, "WriteCapacityUnits": 9}, + "GlobalSecondaryIndexes": [{ + "IndexName": "gi", + "KeySchema": [{"AttributeName": "g", "KeyType": "HASH"}], + "Projection": {"ProjectionType": "KEYS_ONLY"}, + "ProvisionedThroughput": {"ReadCapacityUnits": 3, "WriteCapacityUnits": 4} + }], + "LocalSecondaryIndexes": [{ + "IndexName": "li", + "KeySchema": [ + {"AttributeName": "pk", "KeyType": "HASH"}, + {"AttributeName": "l", "KeyType": "RANGE"} + ], + "Projection": {"ProjectionType": "ALL"} + }] + })) + .expect("input"); + let desc = engine + .create_table_impl(ACCOUNT, input, false) + .await + .expect("create"); + let key_info = TableKeyInfo { + table_name: name.to_owned(), + account_id: ACCOUNT.to_owned(), + table_id: desc.table_id.clone(), + key_schema: desc.key_schema.clone(), + base_key_schema: desc.key_schema.clone(), + attribute_definitions: desc.attribute_definitions.clone(), + ..Default::default() + }; + for i in 0..30 { + let mut item = json!({"pk": {"S": format!("p{}", i % 3)}, "sk": {"N": i.to_string()}}); + if i % 2 == 0 { + item["g"] = json!({"S": format!("g{}", i % 4)}); + } + if i % 3 == 0 { + item["l"] = json!({"S": format!("l{i}")}); + } + let item: Item = serde_json::from_value(item).expect("item"); + engine + .put_item( + &key_info, + item, + false, + None, + &extenddb_core::expression::ExpressionMaps::default(), + None, + ) + .await + .expect("put"); + } + desc.table_id + } + + async fn count(engine: &SqliteEngine, table: &str) -> i64 { + sqlx::query_scalar(&format!("SELECT COUNT(*) FROM {table}")) + .fetch_one(&engine.pool) + .await + .expect("count") + } + + async fn index_tables(engine: &SqliteEngine, table_id: &str) -> Vec<(String, String)> { + sqlx::query_as( + "SELECT index_name, index_id FROM indexes WHERE table_id = ? ORDER BY index_name", + ) + .bind(table_id) + .fetch_all(&engine.pool) + .await + .expect("indexes") + } + + #[tokio::test] + async fn restore_recreates_indexes_and_throughput() { + let engine = engine().await; + let src = indexed_table(&engine, "src").await; + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + let desc = engine + .restore_table_from_backup( + ACCOUNT, + "dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + + let (status, pt): (String, Option) = sqlx::query_as( + "SELECT table_status, provisioned_throughput FROM tables WHERE table_id = ?", + ) + .bind(&desc.table_id) + .fetch_one(&engine.pool) + .await + .expect("row"); + assert_eq!(status, "ACTIVE"); + let pt: serde_json::Value = serde_json::from_str(&pt.expect("throughput")).expect("json"); + assert_eq!(pt["ReadCapacityUnits"], 7); + assert_eq!(pt["WriteCapacityUnits"], 9); + + // Same indexes, each holding the same number of rows as the source's. + let src_idx = index_tables(&engine, &src).await; + let dst_idx = index_tables(&engine, &desc.table_id).await; + assert_eq!( + src_idx.iter().map(|(n, _)| n.as_str()).collect::>(), + vec!["gi", "li"] + ); + assert_eq!( + dst_idx.iter().map(|(n, _)| n.clone()).collect::>(), + src_idx.iter().map(|(n, _)| n.clone()).collect::>() + ); + for ((_, s), (_, d)) in src_idx.iter().zip(&dst_idx) { + let (s, d) = ( + crate::data::index_table_name(s), + crate::data::index_table_name(d), + ); + assert_eq!(count(&engine, &d).await, count(&engine, &s).await); + assert!(count(&engine, &d).await > 0); + } + assert_eq!( + count(&engine, &crate::data::data_table_name(&desc.table_id)).await, + 30 + ); + } + + #[tokio::test] + async fn startup_sweep_removes_an_abandoned_restore() { + let engine = engine().await; + indexed_table(&engine, "src").await; + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + let desc = engine + .restore_table_from_backup( + ACCOUNT, + "dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + let index_ids: Vec = index_tables(&engine, &desc.table_id) + .await + .into_iter() + .map(|(_, id)| id) + .collect(); + + // The state a process killed mid-copy leaves: the target CREATING with + // no scheduled transition (its copy transaction rolled back). + sqlx::query( + "UPDATE tables SET table_status = 'CREATING', status_transition_at = NULL \ + WHERE table_id = ?", + ) + .bind(&desc.table_id) + .execute(&engine.pool) + .await + .expect("simulate crash"); + + let removed = engine.sweep_abandoned_restores().await.expect("sweep"); + assert_eq!(removed, vec!["dst".to_owned()]); + let rows: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM tables WHERE table_name = 'dst'") + .fetch_one(&engine.pool) + .await + .expect("rows"); + assert_eq!(rows, 0); + let mut leftover = vec![desc.table_id.clone()]; + leftover.extend(index_ids); + for id in leftover { + let exists: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?)", + ) + .bind(format!("_ddb_{id}")) + .fetch_one(&engine.pool) + .await + .expect("exists"); + assert!(!exists, "data table _ddb_{id} left behind"); + } + + // A live table is never a candidate, and the name is free again. + assert!( + engine + .sweep_abandoned_restores() + .await + .expect("sweep") + .is_empty() + ); + engine + .restore_table_from_backup( + ACCOUNT, + "dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore again"); + } + + #[tokio::test] + async fn startup_sweep_removes_an_incomplete_backup_and_its_rows() { + let engine = engine().await; + indexed_table(&engine, "src").await; + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + sqlx::query("UPDATE backups SET backup_status = 'CREATING' WHERE backup_arn = ?") + .bind(&backup.backup_arn) + .execute(&engine.pool) + .await + .expect("simulate crash"); + let listed = engine + .list_backups(ACCOUNT, Some("src")) + .await + .expect("list CREATING backup"); + assert_eq!(listed.len(), 1); + assert_eq!(listed[0].backup_status, "CREATING"); + let described = engine + .describe_backup(ACCOUNT, &backup.backup_arn) + .await + .expect("describe CREATING backup"); + assert_eq!(described.backup_details.backup_status, "CREATING"); + engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect_err("DeleteBackup only accepts AVAILABLE backups"); + + let removed = engine + .sweep_incomplete_backups() + .await + .expect("backup sweep"); + assert_eq!(removed, vec![backup.backup_arn.clone()]); + let backup_rows: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM backups WHERE backup_arn = ?") + .bind(&backup.backup_arn) + .fetch_one(&engine.pool) + .await + .expect("backup rows"); + let item_rows: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM backup_items WHERE backup_arn = ?") + .bind(&backup.backup_arn) + .fetch_one(&engine.pool) + .await + .expect("item rows"); + let definition_rows: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM backup_definitions WHERE backup_arn = ?") + .bind(&backup.backup_arn) + .fetch_one(&engine.pool) + .await + .expect("definition rows"); + assert_eq!((backup_rows, item_rows, definition_rows), (0, 0, 0)); + assert!( + engine + .sweep_incomplete_backups() + .await + .expect("second sweep") + .is_empty() + ); + } + + #[tokio::test] + async fn failed_backup_copy_removes_its_creating_row() { + let (engine, path) = file_engine().await; + let source = plain_table(&engine, "source").await; + let table = crate::data::data_table_name(&source.table_id); + sqlx::query(&format!( + "INSERT INTO {table} (pk, item_data) VALUES ('broken', 'not json')" + )) + .execute(&engine.pool) + .await + .expect("corrupt source row"); + + engine + .create_backup(ACCOUNT, "source", "broken") + .await + .expect_err("invalid source item must fail the backup"); + let leftovers: (i64, i64, i64) = sqlx::query_as( + "SELECT \ + (SELECT COUNT(*) FROM backups WHERE backup_name = 'broken'), \ + (SELECT COUNT(*) FROM backup_definitions d \ + JOIN backups b ON b.backup_arn = d.backup_arn WHERE b.backup_name = 'broken'), \ + (SELECT COUNT(*) FROM backup_items i \ + JOIN backups b ON b.backup_arn = i.backup_arn WHERE b.backup_name = 'broken')", + ) + .fetch_one(&engine.pool) + .await + .expect("leftovers"); + assert_eq!(leftovers, (0, 0, 0)); + close_file_engine(engine, &path).await; + } + + #[tokio::test] + async fn failed_copy_removes_the_target() { + let engine = engine().await; + indexed_table(&engine, "src").await; + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + // Corrupt one backup row so the copy fails part-way. + sqlx::query( + "UPDATE backup_items SET item_data = '{\"pk\": {\"S\": \"x\"}}' \ + WHERE id = (SELECT MAX(id) FROM backup_items WHERE backup_arn = ?)", + ) + .bind(&backup.backup_arn) + .execute(&engine.pool) + .await + .expect("corrupt"); + let before: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM sqlite_master WHERE type = 'table'") + .fetch_one(&engine.pool) + .await + .expect("tables"); + engine + .restore_table_from_backup( + ACCOUNT, + "dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect_err("an item without its sort key cannot restore"); + let rows: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM tables WHERE table_name = 'dst'") + .fetch_one(&engine.pool) + .await + .expect("rows"); + assert_eq!(rows, 0); + let after: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM sqlite_master WHERE type = 'table'") + .fetch_one(&engine.pool) + .await + .expect("tables"); + assert_eq!( + after, before, + "the target's data and index tables are dropped" + ); + } + + #[tokio::test] + async fn restore_intent_is_committed_with_target() { + let engine = engine().await; + indexed_table(&engine, "src").await; + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + let input = serde_json::from_value(json!({ + "TableName": "dst", + "KeySchema": [{"AttributeName": "pk", "KeyType": "HASH"}], + "AttributeDefinitions": [{"AttributeName": "pk", "AttributeType": "S"}], + "BillingMode": "PAY_PER_REQUEST" + })) + .expect("input"); + + // This is the state exposed between target creation and the first copy + // batch. Both rows must become visible together, before restore gives + // DeleteBackup any chance to acquire the write lock. + let desc = engine + .create_table_for_restore(ACCOUNT, input, &backup.backup_arn) + .await + .expect("create restore target"); + let mut tx = engine.pool.begin().await.expect("read transaction"); + let state: (bool, bool) = sqlx::query_as( + "SELECT \ + EXISTS(SELECT 1 FROM tables WHERE table_id = ? AND table_status = 'CREATING'), \ + EXISTS(SELECT 1 FROM table_restores \ + WHERE table_id = ? AND source_backup_arn = ?)", + ) + .bind(&desc.table_id) + .bind(&desc.table_id) + .bind(&backup.backup_arn) + .fetch_one(&mut *tx) + .await + .expect("restore state"); + tx.commit().await.expect("read commit"); + assert_eq!(state, (true, true)); + + let err = engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect_err("restore intent must block deletion"); + assert!( + matches!(err, extenddb_storage::error::StorageError::BackupInUse(_)), + "{err:?}" + ); + + assert!(engine.abort_restore(&desc.table_id).await.expect("abort")); + let intents: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM table_restores WHERE source_backup_arn = ?") + .bind(&backup.backup_arn) + .fetch_one(&engine.pool) + .await + .expect("intent count"); + assert_eq!(intents, 0, "target cleanup must remove restore intent"); + engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect("backup is deletable after abort"); + } + + async fn create(engine: &SqliteEngine, input: serde_json::Value) -> String { + let input: extenddb_core::types::CreateTableInput = + serde_json::from_value(input).expect("input"); + engine + .create_table_impl(ACCOUNT, input, false) + .await + .expect("create") + .table_id + } + + #[tokio::test] + async fn multipart_key_backup_is_refused() { + let engine = engine().await; + create( + &engine, + json!({ + "TableName": "m", + "KeySchema": [ + {"AttributeName": "a", "KeyType": "HASH"}, + {"AttributeName": "b", "KeyType": "HASH"} + ], + "AttributeDefinitions": [ + {"AttributeName": "a", "AttributeType": "S"}, + {"AttributeName": "b", "AttributeType": "S"} + ], + "BillingMode": "PAY_PER_REQUEST" + }), + ) + .await; + let backup = engine + .create_backup(ACCOUNT, "m", "bkp") + .await + .expect("backup"); + let err = engine + .restore_table_from_backup( + ACCOUNT, + "m2", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect_err("refused"); + assert!( + matches!(err, extenddb_storage::error::StorageError::Unsupported(_)), + "{err:?}" + ); + let rows: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM tables WHERE table_name = 'm2'") + .fetch_one(&engine.pool) + .await + .expect("rows"); + assert_eq!(rows, 0); + } + + #[tokio::test] + async fn restore_keeps_table_class_sse_and_on_demand_limits() { + let engine = engine().await; + create( + &engine, + json!({ + "TableName": "c", + "KeySchema": [{"AttributeName": "pk", "KeyType": "HASH"}], + "AttributeDefinitions": [{"AttributeName": "pk", "AttributeType": "S"}], + "BillingMode": "PAY_PER_REQUEST", + "TableClass": "STANDARD_INFREQUENT_ACCESS", + "SSESpecification": {"Enabled": true, "SSEType": "KMS"}, + "OnDemandThroughput": {"MaxReadRequestUnits": 100, "MaxWriteRequestUnits": 50} + }), + ) + .await; + let backup = engine + .create_backup(ACCOUNT, "c", "bkp") + .await + .expect("backup"); + engine + .restore_table_from_backup( + ACCOUNT, + "c2", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + let read = |name: &'static str| { + let pool = engine.pool.clone(); + async move { + let row: (String, Option, Option, Option) = sqlx::query_as( + "SELECT billing_mode, table_class, sse_specification, on_demand_throughput \ + FROM tables WHERE table_name = ?", + ) + .bind(name) + .fetch_one(&pool) + .await + .expect("row"); + let json = |v: Option| { + v.map(|s| serde_json::from_str::(&s).expect("json")) + }; + (row.0, row.1, json(row.2), json(row.3)) + } + }; + let want = read("c").await; + assert_eq!(want.1.as_deref(), Some("STANDARD_INFREQUENT_ACCESS")); + assert!(want.2.is_some() && want.3.is_some(), "{want:?}"); + assert_eq!(read("c2").await, want); + } + + #[tokio::test] + async fn restore_crosses_batch_boundaries_and_keeps_n_key_order() { + let engine = engine().await; + let src = indexed_table(&engine, "src").await; + let (src_gsi_before,) = (index_tables(&engine, &src).await[0].1.clone(),); + let base = crate::data::data_table_name(&src); + // Add rows straight into the source, well past one read batch, with N + // sort keys whose encoded order differs from their text order. + let key_info = TableKeyInfo { + table_name: "src".to_owned(), + account_id: ACCOUNT.to_owned(), + table_id: src.clone(), + key_schema: serde_json::from_value(json!([ + {"AttributeName": "pk", "KeyType": "HASH"}, + {"AttributeName": "sk", "KeyType": "RANGE"} + ])) + .expect("ks"), + ..Default::default() + }; + let mut key_info = key_info; + key_info.base_key_schema = key_info.key_schema.clone(); + key_info.attribute_definitions = serde_json::from_value(json!([ + {"AttributeName": "pk", "AttributeType": "S"}, + {"AttributeName": "sk", "AttributeType": "N"}, + {"AttributeName": "g", "AttributeType": "S"}, + {"AttributeName": "l", "AttributeType": "S"} + ])) + .expect("ad"); + // More rows than one read batch (500), so the restore commits + // several batches. + for i in 0..700 { + let n = if i % 2 == 0 { + format!("-{i}.5") + } else { + format!("{}", i * 1000) + }; + let item: Item = serde_json::from_value(json!({ + "pk": {"S": "bulk"}, "sk": {"N": n}, "g": {"S": "gbulk"} + })) + .expect("item"); + engine + .put_item( + &key_info, + item, + false, + None, + &extenddb_core::expression::ExpressionMaps::default(), + None, + ) + .await + .expect("put"); + } + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + let desc = engine + .restore_table_from_backup( + ACCOUNT, + "dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + let dst = crate::data::data_table_name(&desc.table_id); + assert_eq!(count(&engine, &dst).await, count(&engine, &base).await); + assert_eq!(count(&engine, &dst).await, 730); + // The physical rows are identical to the ones the write path made, + // sort-key encoding included, so key order is preserved. + let rows = |t: String| { + let pool = engine.pool.clone(); + async move { + sqlx::query_scalar::<_, String>(&format!( + "SELECT pk || '|' || sk_n || '|' || item_data FROM {t} ORDER BY pk, sk_n" + )) + .fetch_all(&pool) + .await + .expect("rows") + } + }; + assert_eq!(rows(dst).await, rows(base).await); + let dst_gsi = index_tables(&engine, &desc.table_id).await[0].1.clone(); + let gsi_rows = |id: String| { + let pool = engine.pool.clone(); + async move { + sqlx::query_scalar::<_, String>(&format!( + "SELECT pk || '|' || base_pk || '|' || base_sk_n || '|' || item_data \ + FROM {} ORDER BY 1", + crate::data::index_table_name(&id) + )) + .fetch_all(&pool) + .await + .expect("gsi rows") + } + }; + assert_eq!(gsi_rows(dst_gsi).await, gsi_rows(src_gsi_before).await); + } + + /// A database with the schema catalog 0.0.3 shipped, holding a backup in + /// the row shape 0.0.3 wrote (`pk` empty, `sk` NULL, no definition), is + /// brought to 0.0.4 by re-applying the schema, which is what + /// `extenddb migrate` does on this backend; the old backup then restores + /// as it did before: keys and items, no secondary indexes. + #[tokio::test] + async fn a_0_0_3_database_migrates_and_its_backups_restore() { + let engine = SqliteEngine::new(":memory:", 1, "us-east-1", 409_600) + .await + .expect("engine"); + sqlx::raw_sql(include_str!("../testdata/schema_0_0_3.sql")) + .execute(&engine.pool) + .await + .expect("0.0.3 schema"); + sqlx::query("UPDATE settings SET value = '0' WHERE key = 'control_plane_delay_seconds'") + .execute(&engine.pool) + .await + .expect("delay"); + sqlx::query("INSERT INTO accounts (account_id, account_name) VALUES (?, 'default')") + .bind(ACCOUNT) + .execute(&engine.pool) + .await + .expect("account"); + assert!(engine.check_catalog_version().await.is_err()); + + // A backup exactly as the 0.0.3 binary wrote one. + let arn = format!("arn:aws:dynamodb:us-east-1:{ACCOUNT}:table/old/backup/1-00000000"); + sqlx::query( + "INSERT INTO backups (backup_arn, backup_name, table_id, table_name, account_id, \ + backup_status, backup_size_bytes, item_count, key_schema, attribute_definitions, \ + billing_mode) VALUES (?, 'b', 'gone', 'old', ?, 'AVAILABLE', 0, 3, ?, ?, \ + 'PROVISIONED')", + ) + .bind(&arn) + .bind(ACCOUNT) + .bind(r#"[{"AttributeName":"pk","KeyType":"HASH"},{"AttributeName":"sk","KeyType":"RANGE"}]"#) + .bind(r#"[{"AttributeName":"pk","AttributeType":"S"},{"AttributeName":"sk","AttributeType":"N"}]"#) + .execute(&engine.pool) + .await + .expect("old backup row"); + for i in 0..3 { + sqlx::query( + "INSERT INTO backup_items (backup_arn, pk, sk, item_data) VALUES (?, '', NULL, ?)", + ) + .bind(&arn) + .bind(format!( + r#"{{"pk":{{"S":"a"}},"sk":{{"N":"{i}"}},"v":{{"S":"x{i}"}}}}"# + )) + .execute(&engine.pool) + .await + .expect("old backup item"); + } + + crate::schema::apply(&engine.pool).await.expect("migrate"); + engine + .check_catalog_version() + .await + .expect("0.0.4 after migrate"); + let id_is_primary_key: i64 = sqlx::query_scalar( + "SELECT pk FROM pragma_table_info('backup_items') WHERE name = 'id'", + ) + .fetch_one(&engine.pool) + .await + .expect("backup_items id"); + assert_eq!(id_is_primary_key, 1); + let migrated_items: Vec<(i64, String)> = + sqlx::query_as("SELECT id, item_data FROM backup_items ORDER BY id") + .fetch_all(&engine.pool) + .await + .expect("migrated backup items"); + assert_eq!( + migrated_items.iter().map(|(id, _)| *id).collect::>(), + vec![1, 2, 3] + ); + assert!(migrated_items[0].1.contains("x0")); + assert!(migrated_items[1].1.contains("x1")); + assert!(migrated_items[2].1.contains("x2")); + let desc = engine + .restore_table_from_backup(ACCOUNT, "new", &arn, RestoreTableOverrides::default()) + .await + .expect("restore a 0.0.3 backup"); + assert!(index_tables(&engine, &desc.table_id).await.is_empty()); + assert_eq!( + count(&engine, &crate::data::data_table_name(&desc.table_id)).await, + 3 + ); + let pt: String = + sqlx::query_scalar("SELECT provisioned_throughput FROM tables WHERE table_id = ?") + .bind(&desc.table_id) + .fetch_one(&engine.pool) + .await + .expect("throughput"); + let pt: serde_json::Value = serde_json::from_str(&pt).expect("json"); + assert_eq!( + ( + pt["ReadCapacityUnits"].as_i64(), + pt["WriteCapacityUnits"].as_i64() + ), + (Some(5), Some(5)) + ); + } + + #[tokio::test] + async fn batches_are_cut_by_stored_bytes() { + let engine = engine().await; + sqlx::query("CREATE TABLE t (item_data TEXT NOT NULL)") + .execute(&engine.pool) + .await + .expect("table"); + // One row over the budget on its own, then small rows, then two rows + // that together exceed it. + let big = "x".repeat(super::COPY_BATCH_BYTES + 10); + let half = "y".repeat(super::COPY_BATCH_BYTES / 2 + 10); + let mut rows = vec![big.clone()]; + rows.extend((0..10).map(|i| format!("small-{i}"))); + rows.push(half.clone()); + rows.push(half.clone()); + for r in &rows { + sqlx::query("INSERT INTO t (item_data) VALUES (?)") + .bind(r) + .execute(&engine.pool) + .await + .expect("insert"); + } + let mut tx = engine.pool.begin().await.expect("tx"); + let mut last = 0; + let mut seen: Vec = Vec::new(); + let mut batches: Vec = Vec::new(); + loop { + let batch = super::next_batch(&mut tx, "t", "rowid", None, last) + .await + .expect("batch"); + let Some(&(tail, _)) = batch.last() else { + break; + }; + let bytes: usize = batch.iter().map(|(_, d)| d.len()).sum(); + assert!( + batch.len() == 1 || bytes <= super::COPY_BATCH_BYTES, + "{bytes}" + ); + batches.push(batch.len()); + seen.extend(batch.into_iter().map(|(_, d)| d)); + last = tail; + } + assert_eq!(seen, rows, "every row once, in order"); + // The oversized row alone; then small rows and one half; then the other half. + assert_eq!(batches, vec![1, 11, 1]); + } + + #[tokio::test] + async fn delete_table_refuses_a_restore_in_progress() { + let engine = engine().await; + indexed_table(&engine, "src").await; + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + let desc = engine + .restore_table_from_backup( + ACCOUNT, + "dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + // Back to the state the target is in while its copy runs. + sqlx::query( + "UPDATE tables SET table_status = 'CREATING', status_transition_at = NULL \ + WHERE table_id = ?", + ) + .bind(&desc.table_id) + .execute(&engine.pool) + .await + .expect("in progress"); + let err = extenddb_storage::TableEngine::delete_table( + &engine, + ACCOUNT, + extenddb_core::types::DeleteTableInput { + table_name: "dst".to_owned(), + }, + ) + .await + .expect_err("refused"); + assert!( + matches!(err, extenddb_storage::error::StorageError::IndexesInUse(_)), + "{err:?}" + ); + } + + #[tokio::test] + async fn restore_summary_and_backup_in_use() { + use extenddb_storage::TableEngine; + let engine = engine().await; + indexed_table(&engine, "src").await; + let backup = engine + .create_backup(ACCOUNT, "src", "bkp") + .await + .expect("backup"); + let desc = engine + .restore_table_from_backup( + ACCOUNT, + "dst", + &backup.backup_arn, + RestoreTableOverrides::default(), + ) + .await + .expect("restore"); + let describe = |name: &'static str| { + let engine = &engine; + async move { + engine + .describe_table( + ACCOUNT, + extenddb_core::types::DescribeTableInput { + table_name: name.to_owned(), + }, + ) + .await + .expect("describe") + } + }; + let done = describe("dst").await.restore_summary.expect("summary"); + assert_eq!( + done.source_backup_arn.as_deref(), + Some(backup.backup_arn.as_str()) + ); + assert!(!done.restore_in_progress); + assert!(done.restore_date_time > 0.0); + assert!(describe("src").await.restore_summary.is_none()); + + sqlx::query( + "UPDATE tables SET table_status = 'CREATING', status_transition_at = NULL \ + WHERE table_id = ?", + ) + .bind(&desc.table_id) + .execute(&engine.pool) + .await + .expect("in progress"); + assert!( + describe("dst") + .await + .restore_summary + .expect("summary") + .restore_in_progress + ); + let err = engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect_err("in use"); + assert!( + matches!(err, extenddb_storage::error::StorageError::BackupInUse(_)), + "{err:?}" + ); + + sqlx::query("UPDATE tables SET table_status = 'ACTIVE' WHERE table_id = ?") + .bind(&desc.table_id) + .execute(&engine.pool) + .await + .expect("finished"); + engine + .delete_backup(ACCOUNT, &backup.backup_arn) + .await + .expect("deletable once the restore is done"); + let after = describe("dst").await.restore_summary.expect("summary"); + assert_eq!( + after.source_backup_arn.as_deref(), + Some(backup.backup_arn.as_str()) + ); + } + + async fn file_engine() -> (SqliteEngine, std::path::PathBuf) { + file_engine_with_pool_size(4).await + } + + async fn file_engine_with_pool_size(pool_size: u32) -> (SqliteEngine, std::path::PathBuf) { + let target = std::env::var_os("CARGO_TARGET_DIR") + .map(std::path::PathBuf::from) + .unwrap_or_else(|| { + std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../target") + }); + assert!(target.is_dir(), "cargo target directory must exist"); + let path = target.join(format!("backup-test-{}.sqlite", uuid::Uuid::new_v4())); + let engine = SqliteEngine::new( + path.to_str().expect("UTF-8 path"), + pool_size, + "us-east-1", + 409_600, + ) + .await + .expect("file engine"); + crate::schema::apply(&engine.pool).await.expect("schema"); + sqlx::query("UPDATE settings SET value = '0' WHERE key = 'control_plane_delay_seconds'") + .execute(&engine.pool) + .await + .expect("delay"); + sqlx::query("UPDATE settings SET value = '0' WHERE key = 'index_propagation_delay_ms'") + .execute(&engine.pool) + .await + .expect("propagation"); + sqlx::query("INSERT INTO accounts (account_id, account_name) VALUES (?, 'default')") + .bind(ACCOUNT) + .execute(&engine.pool) + .await + .expect("account"); + (engine, path) + } + + async fn close_file_engine(engine: SqliteEngine, path: &std::path::Path) { + engine.pool.close().await; + drop(engine); + for suffix in ["", "-wal", "-shm"] { + let _ = std::fs::remove_file(format!("{}{suffix}", path.display())); + } + } + + async fn plain_table(engine: &SqliteEngine, name: &str) -> TableKeyInfo { + let table_id = create( + engine, + json!({ + "TableName": name, + "KeySchema": [{"AttributeName": "pk", "KeyType": "HASH"}], + "AttributeDefinitions": [{"AttributeName": "pk", "AttributeType": "S"}], + "BillingMode": "PAY_PER_REQUEST" + }), + ) + .await; + let key_schema: Vec = + serde_json::from_value(json!([ + {"AttributeName": "pk", "KeyType": "HASH"} + ])) + .expect("key schema"); + TableKeyInfo { + table_name: name.to_owned(), + account_id: ACCOUNT.to_owned(), + table_id, + key_schema: key_schema.clone(), + base_key_schema: key_schema, + attribute_definitions: serde_json::from_value(json!([ + {"AttributeName": "pk", "AttributeType": "S"} + ])) + .expect("attribute definitions"), + ..Default::default() + } + } + + async fn fill_table(engine: &SqliteEngine, key_info: &TableKeyInfo, count: i64) { + let table = crate::data::data_table_name(&key_info.table_id); + let sql = format!( + "WITH RECURSIVE n(i) AS (SELECT 0 UNION ALL SELECT i + 1 FROM n WHERE i + 1 < ?) \ + INSERT INTO {table} (pk, item_data) \ + SELECT 'p' || i, json_object('pk', json_object('S', 'p' || i), \ + 'payload', json_object('S', ?)) FROM n" + ); + let payload = "x".repeat(1024); + let _writer = engine.write_lock.lock().await; + let mut tx = engine + .pool + .begin_with("BEGIN IMMEDIATE") + .await + .expect("write transaction"); + sqlx::query(&sql) + .bind(count) + .bind(payload) + .execute(&mut *tx) + .await + .expect("fill table"); + tx.commit().await.expect("commit fill"); + } + + #[tokio::test] + async fn backup_batches_do_not_stall_an_unrelated_writer() { + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{Arc, Mutex}; + use std::time::Instant; + + let (engine, path) = file_engine_with_pool_size(1).await; + assert!(!engine.in_memory); + assert_eq!( + engine.pool.options().get_max_connections(), + 2, + "file backups require concurrent WAL reader and batch writer connections" + ); + let source = plain_table(&engine, "source").await; + fill_table(&engine, &source, 5_000).await; + let writer_key = plain_table(&engine, "writer").await; + + let stop = Arc::new(AtomicBool::new(false)); + let completions = Arc::new(Mutex::new(Vec::new())); + let task_engine = engine.clone(); + let task_stop = Arc::clone(&stop); + let task_completions = Arc::clone(&completions); + let writer = tokio::spawn(async move { + let mut i = 0_u64; + while !task_stop.load(Ordering::Relaxed) { + let item: Item = serde_json::from_value(json!({ + "pk": {"S": format!("w{i}")} + })) + .expect("writer item"); + task_engine + .put_item( + &writer_key, + item, + false, + None, + &extenddb_core::expression::ExpressionMaps::default(), + None, + ) + .await + .expect("writer put"); + task_completions + .lock() + .expect("completion lock") + .push(Instant::now()); + i += 1; + } + }); + while completions.lock().expect("completion lock").len() < 20 { + tokio::task::yield_now().await; + } + + let started = Instant::now(); + engine + .create_backup(ACCOUNT, "source", "concurrent") + .await + .expect("backup"); + let finished = Instant::now(); + while completions + .lock() + .expect("completion lock") + .last() + .is_none_or(|t| *t <= finished) + { + tokio::task::yield_now().await; + } + stop.store(true, Ordering::Relaxed); + writer.await.expect("writer task"); + + { + let times = completions.lock().expect("completion lock"); + assert!( + times.iter().any(|t| *t > started && *t < finished), + "the unrelated writer must make progress while the backup runs" + ); + let backup_duration = finished.duration_since(started); + let max_gap = times + .windows(2) + .filter(|pair| pair[0] <= finished && pair[1] >= started) + .map(|pair| pair[1].duration_since(pair[0])) + .max() + .expect("writer gaps across backup"); + assert!( + max_gap < backup_duration.mul_f64(0.75), + "writer gap {max_gap:?} must be well below backup duration {backup_duration:?}" + ); + } + close_file_engine(engine, &path).await; + } + + #[tokio::test] + async fn backup_excludes_items_written_after_its_snapshot() { + let (engine, path) = file_engine().await; + let source = plain_table(&engine, "source").await; + fill_table(&engine, &source, 5_000).await; + + let backup_engine = engine.clone(); + let backup = tokio::spawn(async move { + backup_engine + .create_backup(ACCOUNT, "source", "snapshot") + .await + }); + loop { + let status: Option = sqlx::query_scalar( + "SELECT backup_status FROM backups WHERE backup_name = 'snapshot'", + ) + .fetch_optional(&engine.pool) + .await + .expect("backup status"); + match status.as_deref() { + Some("CREATING") => break, + Some("AVAILABLE") => panic!("backup completed before the concurrent write"), + None => tokio::task::yield_now().await, + Some(other) => panic!("unexpected backup status {other}"), + } + } + + let late_item: Item = serde_json::from_value(json!({ + "pk": {"S": "late"}, "payload": {"S": "after snapshot"} + })) + .expect("late item"); + engine + .put_item( + &source, + late_item, + false, + None, + &extenddb_core::expression::ExpressionMaps::default(), + None, + ) + .await + .expect("late put"); + assert!( + !backup.is_finished(), + "the source write must complete while backup copying continues" + ); + let details = backup.await.expect("backup task").expect("backup"); + + let backed_up: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM backup_items WHERE backup_arn = ?") + .bind(&details.backup_arn) + .fetch_one(&engine.pool) + .await + .expect("backup count"); + assert_eq!(backed_up, 5_000); + let late_backed_up: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM backup_items \ + WHERE backup_arn = ? AND item_data LIKE '%after snapshot%')", + ) + .bind(&details.backup_arn) + .fetch_one(&engine.pool) + .await + .expect("late item lookup"); + assert!(!late_backed_up); + let empty_keys: i64 = sqlx::query_scalar( + "SELECT COUNT(*) FROM backup_items WHERE backup_arn = ? AND pk = ''", + ) + .bind(&details.backup_arn) + .fetch_one(&engine.pool) + .await + .expect("backup keys"); + assert_eq!(empty_keys, 0, "backup rows must retain their primary keys"); + close_file_engine(engine, &path).await; + } + + #[tokio::test] + async fn cancelled_create_backup_finishes_with_all_items() { + let (engine, path) = file_engine_with_pool_size(2).await; + let source = plain_table(&engine, "source").await; + const SOURCE_ITEMS: i64 = 10_000; + fill_table(&engine, &source, SOURCE_ITEMS).await; + + let request_engine = engine.clone(); + let request = tokio::spawn(async move { + request_engine + .create_backup(ACCOUNT, "source", "cancelled-request") + .await + }); + tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + let status: Option = sqlx::query_scalar( + "SELECT backup_status FROM backups WHERE backup_name = 'cancelled-request'", + ) + .fetch_optional(&engine.pool) + .await + .expect("backup status"); + match status.as_deref() { + Some("CREATING") => break, + Some(other) => panic!("unexpected backup status before cancellation: {other}"), + None => tokio::task::yield_now().await, + } + } + }) + .await + .expect("backup must enter CREATING"); + + // This drops the request future that is awaiting CreateBackup's inner + // JoinHandle. The detached copy task must remain alive. + request.abort(); + assert!( + request + .await + .expect_err("request was cancelled") + .is_cancelled() + ); + + let (backup_arn, item_count): (String, i64) = + tokio::time::timeout(std::time::Duration::from_secs(30), async { + loop { + let row: Option<(String, String, i64)> = sqlx::query_as( + "SELECT backup_arn, backup_status, item_count FROM backups \ + WHERE backup_name = 'cancelled-request'", + ) + .fetch_optional(&engine.pool) + .await + .expect("backup row"); + match row { + Some((arn, status, count)) if status == "AVAILABLE" => break (arn, count), + Some((_, status, _)) if status == "CREATING" => { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + Some((_, status, _)) => panic!("unexpected final backup status: {status}"), + None => panic!("detached backup disappeared"), + } + } + }) + .await + .expect("detached backup must leave CREATING"); + assert_eq!(item_count, SOURCE_ITEMS); + let copied: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM backup_items WHERE backup_arn = ?") + .bind(&backup_arn) + .fetch_one(&engine.pool) + .await + .expect("copied item count"); + assert_eq!(copied, SOURCE_ITEMS); + + close_file_engine(engine, &path).await; + } +} diff --git a/crates/storage-sqlite/src/config.rs b/crates/storage-sqlite/src/config.rs index 771969dab..84246faea 100644 --- a/crates/storage-sqlite/src/config.rs +++ b/crates/storage-sqlite/src/config.rs @@ -11,8 +11,9 @@ use serde::Deserialize; /// SQLite backend configuration. /// /// `path` is the database file location; `:memory:` selects an ephemeral -/// in-memory database. `pool_size` bounds the read connection pool (writes are -/// serialized by the engine regardless). +/// in-memory database. `pool_size` bounds the connection pool (writes are +/// serialized by the engine regardless). File databases clamp this to at least +/// two connections so a backup reader can coexist with its batched writer. #[derive(Debug, Clone, Deserialize)] #[serde(deny_unknown_fields)] pub struct SqliteConfig { diff --git a/crates/storage-sqlite/src/create_table.rs b/crates/storage-sqlite/src/create_table.rs index 0ff6d8a9a..975d5fbb8 100644 --- a/crates/storage-sqlite/src/create_table.rs +++ b/crates/storage-sqlite/src/create_table.rs @@ -34,6 +34,29 @@ impl SqliteEngine { account_id: &str, input: CreateTableInput, defer_active: bool, + ) -> Result { + self.create_table_impl_inner(account_id, input, defer_active, None) + .await + } + + /// Create a restore target and record its source while both catalog rows + /// are protected by the same write transaction. + pub(crate) async fn create_table_for_restore( + &self, + account_id: &str, + input: CreateTableInput, + backup_arn: &str, + ) -> Result { + self.create_table_impl_inner(account_id, input, true, Some(backup_arn)) + .await + } + + async fn create_table_impl_inner( + &self, + account_id: &str, + input: CreateTableInput, + defer_active: bool, + restore_source_backup_arn: Option<&str>, ) -> Result { Self::validate_account_id(account_id)?; // D1: every writer holds the engine write lock. This method runs two @@ -116,6 +139,23 @@ impl SqliteEngine { .await .map_err(|e| StorageError::Internal(e.to_string()))?; + if let Some(backup_arn) = restore_source_backup_arn { + let available: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM backups WHERE backup_arn = ? \ + AND account_id = ? AND backup_status = 'AVAILABLE')", + ) + .bind(backup_arn) + .bind(account_id) + .fetch_one(&mut *tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + if !available { + return Err(StorageError::Validation(format!( + "Backup not found: {backup_arn}" + ))); + } + } + sqlx::query( "INSERT INTO tables \ (account_id, table_name, key_schema, attribute_definitions, billing_mode, \ @@ -150,6 +190,15 @@ impl SqliteEngine { } })?; + if let Some(backup_arn) = restore_source_backup_arn { + sqlx::query("INSERT INTO table_restores (table_id, source_backup_arn) VALUES (?, ?)") + .bind(&table_id) + .bind(backup_arn) + .execute(&mut *tx) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + } + // GSI / LSI metadata. let mut gsi_ids: Vec = Vec::new(); if let Some(gsis) = &input.global_secondary_indexes { diff --git a/crates/storage-sqlite/src/data/mod.rs b/crates/storage-sqlite/src/data/mod.rs index 20d5db839..b1be9b6e2 100644 --- a/crates/storage-sqlite/src/data/mod.rs +++ b/crates/storage-sqlite/src/data/mod.rs @@ -42,9 +42,9 @@ mod update_item; pub(crate) mod vector_index; pub(crate) use index::{ - PendingApplyContext, apply_pending_context, insert_index_row_multi, project_item_for_index, + PendingApplyContext, apply_pending_context, insert_index_row_multi, item_has_index_keys, + project_item_for_index, }; -pub(crate) use tx_helpers::upsert_item_in_tx; /// Quoted SQL identifier for a virtual DynamoDB table's data table. pub(crate) fn data_table_name(table_id: &str) -> String { diff --git a/crates/storage-sqlite/src/delete_table.rs b/crates/storage-sqlite/src/delete_table.rs index 4621b387d..72533abea 100644 --- a/crates/storage-sqlite/src/delete_table.rs +++ b/crates/storage-sqlite/src/delete_table.rs @@ -37,6 +37,29 @@ impl SqliteEngine { return Err(StorageError::DeletionProtected(row.table_arn.clone())); } + // A restore target (CREATING with no scheduled transition) is being + // filled by RestoreTableFromBackup. The service refuses to delete a + // table it is still creating, and deleting this one would make the + // restore fail part-way; refuse until the restore finishes. An + // abandoned target is removed by the restore sweep, not here. + if row.table_status == "CREATING" { + let restoring: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM tables WHERE table_id = ? \ + AND table_status = 'CREATING' AND status_transition_at IS NULL)", + ) + .bind(&row.table_id) + .fetch_one(&self.pool) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + if restoring { + return Err(StorageError::IndexesInUse(format!( + "Attempt to change a resource which is still in use: Table is being \ + restored: {}", + row.table_name + ))); + } + } + let index_rows: Vec = sqlx::query_as(&format!( "SELECT {INDEX_COLUMNS} FROM indexes WHERE table_id = ?" )) diff --git a/crates/storage-sqlite/src/lib.rs b/crates/storage-sqlite/src/lib.rs index 87427b883..78f4befaa 100644 --- a/crates/storage-sqlite/src/lib.rs +++ b/crates/storage-sqlite/src/lib.rs @@ -213,9 +213,7 @@ fn sqlite_server_components_factory( // works with a persistent path too; every bootstrap step is guarded // (IF NOT EXISTS / INSERT OR IGNORE), so re-running on an initialized // file is a no-op. Production builds keep the explicit-`init` contract. - let in_memory = db_path == ":memory:" - || db_path.starts_with("file::memory:") - || db_path.contains("mode=memory"); + let in_memory = engine.in_memory; if in_memory || options.bootstrap_if_uninitialized { let admin_user = std::env::var("EXTENDDB_ADMIN_USER").ok(); let admin_password = std::env::var("EXTENDDB_ADMIN_PASSWORD").ok(); @@ -251,6 +249,33 @@ fn sqlite_server_components_factory( Err(e) => tracing::error!("Failed to recover control plane transitions: {e}"), } + // Remove any restore target a prior crash left mid-copy. Before the + // server takes requests, so no restore can be in flight. + match engine.sweep_abandoned_restores().await { + Ok(names) => { + for name in &names { + tracing::warn!( + "removed table '{name}': its restore did not finish before the last \ + shutdown" + ); + } + } + Err(e) => tracing::error!("Failed to sweep abandoned restores: {e}"), + } + + // Likewise a backup a prior crash left CREATING: its items were only + // partly copied, and nothing will finish them. + match engine.sweep_incomplete_backups().await { + Ok(arns) => { + for arn in &arns { + tracing::warn!( + "removed backup {arn}: its copy did not finish before the last shutdown" + ); + } + } + Err(e) => tracing::error!("Failed to sweep incomplete backups: {e}"), + } + // Rebuild any GSI left mid-backfill (status CREATING) by a prior crash. match engine.reconcile_incomplete_gsis().await { Ok(n) if n > 0 => tracing::info!("Reconciled {n} incomplete GSI(s) at startup"), diff --git a/crates/storage-sqlite/src/schema.rs b/crates/storage-sqlite/src/schema.rs index f28532e8e..c1b52668c 100644 --- a/crates/storage-sqlite/src/schema.rs +++ b/crates/storage-sqlite/src/schema.rs @@ -27,7 +27,7 @@ use sqlx::SqlitePool; /// Compiled-in catalog version. Single source of truth for the SQLite backend; /// mirrors the PostgreSQL backend's `CATALOG_VERSION`. pub const CATALOG_VERSION: extenddb_core::version::CatalogVersion = - extenddb_core::version::CatalogVersion::new(0, 0, 3); + extenddb_core::version::CatalogVersion::new(0, 0, 4); /// Complete catalog schema, applied once on a fresh database. /// @@ -365,8 +365,10 @@ CREATE TABLE IF NOT EXISTS backups ( CREATE INDEX IF NOT EXISTS idx_backups_table ON backups (account_id, table_name); --- Backup items. +-- Backup items. `id` is the stable restore cursor: unlike an implicit rowid, +-- an INTEGER PRIMARY KEY value is not renumbered by VACUUM. CREATE TABLE IF NOT EXISTS backup_items ( + id INTEGER PRIMARY KEY AUTOINCREMENT, backup_arn TEXT NOT NULL REFERENCES backups(backup_arn) ON DELETE CASCADE, pk TEXT NOT NULL, sk TEXT, @@ -375,6 +377,26 @@ CREATE TABLE IF NOT EXISTS backup_items ( CREATE INDEX IF NOT EXISTS idx_backup_items_arn ON backup_items (backup_arn); +-- A backup's table definition beyond its keys: secondary indexes, billing +-- mode and throughput, table class, encryption (catalog 0.0.4). JSON in the +-- wire's shape behind a version marker; see +-- `extenddb_storage::backup_definition`. A backup taken before 0.0.4 has no +-- row and restores as before. +CREATE TABLE IF NOT EXISTS backup_definitions ( + backup_arn TEXT PRIMARY KEY REFERENCES backups(backup_arn) ON DELETE CASCADE, + definition TEXT NOT NULL +); + +-- Which backup a restored table came from, and when (catalog 0.0.4): reported +-- by DescribeTable as RestoreSummary, and used to refuse DeleteBackup while the +-- restore is running. Not a foreign key to backups, which may be deleted later. +CREATE TABLE IF NOT EXISTS table_restores ( + table_id TEXT PRIMARY KEY REFERENCES tables(table_id) ON DELETE CASCADE, + source_backup_arn TEXT NOT NULL, + restore_date_time TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')) +); +CREATE INDEX IF NOT EXISTS idx_table_restores_backup ON table_restores (source_backup_arn); + -- Continuous backups / PITR status. CREATE TABLE IF NOT EXISTS continuous_backups ( account_id TEXT NOT NULL, @@ -426,7 +448,7 @@ INSERT OR IGNORE INTO seq_counters (name, value) -- never advance, so a migration could add objects and still leave the server -- refusing to start on a version mismatch. Must stay in step with -- `CATALOG_VERSION` above; they are checked against each other in a test. -INSERT INTO settings (key, value) VALUES ('catalog_version', '0.0.3') +INSERT INTO settings (key, value) VALUES ('catalog_version', '0.0.4') ON CONFLICT(key) DO UPDATE SET value = excluded.value; INSERT OR IGNORE INTO settings (key, value) VALUES ('control_plane_delay_seconds', '0.25'); INSERT OR IGNORE INTO settings (key, value) VALUES ('index_propagation_delay_ms', '10'); @@ -436,6 +458,32 @@ INSERT OR IGNORE INTO settings (key, value) VALUES ('index_propagation_delay_ms' /// /// Idempotent: every statement uses `IF NOT EXISTS` / `INSERT OR IGNORE`. pub async fn apply(pool: &SqlitePool) -> OpResult<()> { + // The schema's last statement stamps the compiled version. Never stamp an + // older version onto a catalog a newer binary has already upgraded: that + // binary's startup gate would keep passing and this one's would start + // passing against tables it does not know. `extenddb migrate` refuses such + // a catalog before getting here; this guards every other caller. + if table_exists(pool, "settings").await? { + let stored: Option<(String,)> = + sqlx::query_as("SELECT value FROM settings WHERE key = 'catalog_version'") + .fetch_optional(pool) + .await + .map_err(|e| OpError::Internal(format!("read catalog_version: {e}")))?; + if let Some((stored,)) = stored + && let Ok(stored_v) = stored.parse::() + && stored_v > CATALOG_VERSION + { + return Err(OpError::Internal(format!( + "catalog version {stored} is newer than this binary's {CATALOG_VERSION}; \ + not downgrading it" + ))); + } + } + // Catalog 0.0.4 originally created backup_items without a stable cursor. + // Rebuild it before SCHEMA_SQL stamps 0.0.4, including databases that an + // earlier 0.0.4-in-progress binary already stamped. SQLite cannot add an + // INTEGER PRIMARY KEY with ALTER TABLE. + migrate_backup_items(pool).await?; sqlx::raw_sql(SCHEMA_SQL) .execute(pool) .await @@ -443,6 +491,66 @@ pub async fn apply(pool: &SqlitePool) -> OpResult<()> { Ok(()) } +async fn migrate_backup_items(pool: &SqlitePool) -> OpResult<()> { + if !table_exists(pool, "backup_items").await? { + return Ok(()); + } + let has_id: bool = sqlx::query_scalar( + "SELECT EXISTS(SELECT 1 FROM pragma_table_info('backup_items') WHERE name = 'id')", + ) + .fetch_one(pool) + .await + .map_err(|e| OpError::Internal(format!("inspect backup_items columns: {e}")))?; + if has_id { + return Ok(()); + } + + let mut tx = pool + .begin() + .await + .map_err(|e| OpError::Internal(format!("begin backup_items migration: {e}")))?; + sqlx::query("DROP TABLE IF EXISTS backup_items_new") + .execute(&mut *tx) + .await + .map_err(|e| OpError::Internal(format!("remove stale backup_items_new: {e}")))?; + sqlx::query( + "CREATE TABLE backup_items_new (\ + id INTEGER PRIMARY KEY AUTOINCREMENT, \ + backup_arn TEXT NOT NULL REFERENCES backups(backup_arn) ON DELETE CASCADE, \ + pk TEXT NOT NULL, sk TEXT, item_data TEXT NOT NULL)", + ) + .execute(&mut *tx) + .await + .map_err(|e| OpError::Internal(format!("create backup_items_new: {e}")))?; + sqlx::query( + "INSERT INTO backup_items_new (backup_arn, pk, sk, item_data) \ + SELECT backup_arn, pk, sk, item_data FROM backup_items ORDER BY rowid", + ) + .execute(&mut *tx) + .await + .map_err(|e| OpError::Internal(format!("copy backup_items: {e}")))?; + sqlx::query("DROP INDEX IF EXISTS idx_backup_items_arn") + .execute(&mut *tx) + .await + .map_err(|e| OpError::Internal(format!("drop old backup_items index: {e}")))?; + sqlx::query("DROP TABLE backup_items") + .execute(&mut *tx) + .await + .map_err(|e| OpError::Internal(format!("drop old backup_items: {e}")))?; + sqlx::query("ALTER TABLE backup_items_new RENAME TO backup_items") + .execute(&mut *tx) + .await + .map_err(|e| OpError::Internal(format!("rename backup_items_new: {e}")))?; + sqlx::query("CREATE INDEX idx_backup_items_arn ON backup_items (backup_arn)") + .execute(&mut *tx) + .await + .map_err(|e| OpError::Internal(format!("recreate backup_items index: {e}")))?; + tx.commit() + .await + .map_err(|e| OpError::Internal(format!("commit backup_items migration: {e}")))?; + Ok(()) +} + /// Return whether a table with the given name exists in the catalog. pub async fn table_exists(pool: &SqlitePool, name: &str) -> OpResult { let exists: bool = sqlx::query_scalar( @@ -487,3 +595,92 @@ mod tests { ); } } + +#[cfg(test)] +mod apply_tests { + use super::{CATALOG_VERSION, apply}; + use sqlx::sqlite::SqlitePoolOptions; + + #[tokio::test] + async fn a_newer_catalog_is_not_stamped_down() { + let pool = SqlitePoolOptions::new() + .max_connections(1) + .connect("sqlite::memory:") + .await + .expect("pool"); + apply(&pool).await.expect("fresh apply"); + let newer = format!("{}.{}.{}", 0, 0, 999); + sqlx::query("UPDATE settings SET value = ? WHERE key = 'catalog_version'") + .bind(&newer) + .execute(&pool) + .await + .expect("stamp newer"); + let err = apply(&pool).await.expect_err("re-apply on a newer catalog"); + assert!( + format!("{err:?}").contains("newer than this binary"), + "{err:?}" + ); + let (stored,): (String,) = + sqlx::query_as("SELECT value FROM settings WHERE key = 'catalog_version'") + .fetch_one(&pool) + .await + .expect("read"); + assert_eq!(stored, newer, "stored version must be untouched"); + assert!(stored != CATALOG_VERSION.to_string()); + } + + #[tokio::test] + async fn backup_items_migration_is_idempotent_and_preserves_row_order() { + let pool = SqlitePoolOptions::new() + .max_connections(1) + .connect("sqlite::memory:") + .await + .expect("pool"); + sqlx::raw_sql( + "CREATE TABLE backups (\ + backup_arn TEXT PRIMARY KEY, account_id TEXT NOT NULL, table_name TEXT NOT NULL); \ + CREATE TABLE backup_items (\ + backup_arn TEXT NOT NULL REFERENCES backups(backup_arn) ON DELETE CASCADE, \ + pk TEXT NOT NULL, sk TEXT, item_data TEXT NOT NULL); \ + CREATE INDEX idx_backup_items_arn ON backup_items (backup_arn); \ + INSERT INTO backups (backup_arn, account_id, table_name) VALUES ('a', '0', 't'); \ + INSERT INTO backup_items (backup_arn, pk, sk, item_data) VALUES \ + ('a', 'z', NULL, 'first'), \ + ('a', 'a', NULL, 'second'), \ + ('a', 'm', NULL, 'third');", + ) + .execute(&pool) + .await + .expect("old schema"); + + apply(&pool).await.expect("first migration"); + let rows: Vec<(i64, String, String)> = + sqlx::query_as("SELECT id, pk, item_data FROM backup_items ORDER BY id") + .fetch_all(&pool) + .await + .expect("migrated rows"); + assert_eq!( + rows, + vec![ + (1, "z".to_owned(), "first".to_owned()), + (2, "a".to_owned(), "second".to_owned()), + (3, "m".to_owned(), "third".to_owned()), + ] + ); + + apply(&pool).await.expect("second migration"); + let after: Vec<(i64, String, String)> = + sqlx::query_as("SELECT id, pk, item_data FROM backup_items ORDER BY id") + .fetch_all(&pool) + .await + .expect("rows after second apply"); + assert_eq!(after, rows); + let primary_key: i64 = sqlx::query_scalar( + "SELECT pk FROM pragma_table_info('backup_items') WHERE name = 'id'", + ) + .fetch_one(&pool) + .await + .expect("id column"); + assert_eq!(primary_key, 1); + } +} diff --git a/crates/storage-sqlite/src/store.rs b/crates/storage-sqlite/src/store.rs index 995ed7397..90b7d9cf6 100644 --- a/crates/storage-sqlite/src/store.rs +++ b/crates/storage-sqlite/src/store.rs @@ -55,6 +55,9 @@ pub struct SqliteEngine { pub(crate) pool: SqlitePool, pub(crate) region: String, pub(crate) max_item_size_bytes: usize, + /// Whether this engine owns a single-connection in-memory database. + /// Backup strategy depends on this property, not on user pool sizing. + pub(crate) in_memory: bool, /// Wakes the control-plane poller when a table enters CREATING / DELETING. pub(crate) control_plane_notify: Arc, /// Cached default GSI propagation delay (ms); refreshed by a worker and @@ -91,17 +94,19 @@ impl SqliteEngine { max_item_size_bytes: usize, ) -> Result { let url = sqlite_url(path_or_url); + // Use the same parsed-location classifier as the serve lock. String + // containment misclassifies file names such as `mode=memory.db`. + let in_memory = crate::serve_lock::database_file(path_or_url) + .map_err(StorageError::Connection)? + .is_none(); // An in-memory database lives only inside its own connection: a second - // connection opens a *separate* empty database. So for `:memory:` we pin - // the pool to a single connection that is never recycled (idle and - // lifetime timeouts disabled), guaranteeing one shared database for the - // process lifetime. Writes are already serialized by `write_lock`, so a - // single connection costs only read concurrency — acceptable for the - // ephemeral in-memory use case. File-backed databases keep the full WAL - // pool (concurrent readers alongside a single writer). - let in_memory = url.contains(":memory:") || url.contains("mode=memory"); + // connection opens a *separate* empty database. So for an in-memory + // location we pin the pool to a single connection that is never + // recycled. File-backed backups hold a WAL reader while committing + // bounded backup-item batches, so they require a second connection + // even when `pool_size = 1` was configured. let mut opts = SqlitePoolOptions::new() - .max_connections(if in_memory { 1 } else { pool_size.max(1) }) + .max_connections(if in_memory { 1 } else { pool_size.max(2) }) .min_connections(1); if in_memory { opts = opts.idle_timeout(None).max_lifetime(None); @@ -151,6 +156,7 @@ impl SqliteEngine { pool, region: region.to_owned(), max_item_size_bytes, + in_memory, control_plane_notify: Arc::new(tokio::sync::Notify::new()), index_propagation_delay_cache: Arc::new(AtomicU64::new(initial_index_delay)), gsi_notify: Arc::new(tokio::sync::Notify::new()), diff --git a/crates/storage-sqlite/src/table_helpers.rs b/crates/storage-sqlite/src/table_helpers.rs index b8adfbf1e..7554565ca 100644 --- a/crates/storage-sqlite/src/table_helpers.rs +++ b/crates/storage-sqlite/src/table_helpers.rs @@ -150,7 +150,9 @@ impl SqliteEngine { .map_err(|e| StorageError::Internal(e.to_string()))?; let table_name_owned = row.table_name.clone(); + let table_id = row.table_id.clone(); let mut desc = self.build_table_description_from_row(account_id, row, index_rows)?; + desc.restore_summary = self.restore_summary(&table_id, &desc.table_status).await?; let catalog_rows = vector_rows .into_iter() .map(VectorIndexRow::into_catalog_row) @@ -388,3 +390,32 @@ fn zero_throughput() -> ProvisionedThroughputDescription { last_decrease_date_time: None, } } + +impl SqliteEngine { + /// The RestoreSummary of a table created by RestoreTableFromBackup, or + /// `None` for any other table. In progress until the table is ACTIVE. + pub(crate) async fn restore_summary( + &self, + table_id: &str, + status: &extenddb_core::types::TableStatus, + ) -> Result, StorageError> { + let row: Option<(String, String)> = sqlx::query_as( + "SELECT source_backup_arn, restore_date_time FROM table_restores WHERE table_id = ?", + ) + .bind(table_id) + .fetch_optional(&self.pool) + .await + .map_err(|e| StorageError::Internal(e.to_string()))?; + Ok(row.map(|(arn, at)| { + #[allow(clippy::cast_precision_loss)] + let at = crate::sqlite_util::parse_timestamp(&at) + .map(|t| t.unix_timestamp_nanos() as f64 / 1e9) + .unwrap_or(0.0); + extenddb_core::types::RestoreSummary { + source_backup_arn: Some(arn), + restore_date_time: at, + restore_in_progress: *status == extenddb_core::types::TableStatus::Creating, + } + })) + } +} diff --git a/crates/storage-sqlite/testdata/schema_0_0_3.sql b/crates/storage-sqlite/testdata/schema_0_0_3.sql new file mode 100644 index 000000000..172ffbaf8 --- /dev/null +++ b/crates/storage-sqlite/testdata/schema_0_0_3.sql @@ -0,0 +1,400 @@ +-- Copyright 2026 ExtendDB contributors +-- SPDX-License-Identifier: Apache-2.0 +-- The SQLite catalog schema exactly as catalog 0.0.3 shipped it (main at +-- c178814), for the upgrade test in src/backup.rs. Do not edit. + +-- Accounts — multi-account support (REQ-AUTH-005). +CREATE TABLE IF NOT EXISTS accounts ( + account_id TEXT PRIMARY KEY, + account_name TEXT NOT NULL UNIQUE, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')) +); + +-- Table metadata. +CREATE TABLE IF NOT EXISTS tables ( + account_id TEXT NOT NULL REFERENCES accounts(account_id) ON DELETE CASCADE, + table_name TEXT NOT NULL, + key_schema TEXT NOT NULL, + attribute_definitions TEXT NOT NULL, + billing_mode TEXT NOT NULL DEFAULT 'PAY_PER_REQUEST', + provisioned_throughput TEXT, + stream_specification TEXT, + table_status TEXT NOT NULL DEFAULT 'CREATING', + creation_date_time TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + table_size_bytes INTEGER NOT NULL DEFAULT 0, + item_count INTEGER NOT NULL DEFAULT 0, + table_arn TEXT NOT NULL, + table_id TEXT NOT NULL, + ttl_attribute TEXT, + deletion_protection_enabled INTEGER NOT NULL DEFAULT 0, + status_transition_at TEXT, + stream_label TEXT, + ttl_index_ready INTEGER NOT NULL DEFAULT 0, + table_class TEXT, + sse_specification TEXT, + on_demand_throughput TEXT, + PRIMARY KEY (account_id, table_name), + CONSTRAINT tables_table_id_unique UNIQUE (table_id) +); + +CREATE INDEX IF NOT EXISTS idx_tables_pending_transition + ON tables (status_transition_at) + WHERE status_transition_at IS NOT NULL; + +-- Index metadata. index_id is always supplied by the engine (no DB-side UUID). +CREATE TABLE IF NOT EXISTS indexes ( + table_id TEXT NOT NULL, + index_id TEXT NOT NULL, + index_name TEXT NOT NULL, + index_type TEXT NOT NULL, + key_schema TEXT NOT NULL, + projection TEXT NOT NULL, + index_status TEXT NOT NULL DEFAULT 'ACTIVE', + provisioned_throughput TEXT, + propagation_delay_ms INTEGER, + PRIMARY KEY (table_id, index_name), + CONSTRAINT indexes_table_id_fkey + FOREIGN KEY (table_id) REFERENCES tables(table_id) ON DELETE CASCADE, + CONSTRAINT chk_propagation_delay_ms_non_negative + CHECK (propagation_delay_ms IS NULL OR propagation_delay_ms >= 0) +); + +-- Vector index metadata. Kept out of `indexes` deliberately: a vector index is +-- not described by a key schema, so reusing that table's `key_schema` column +-- would mean storing something meaningless in a NOT NULL column. The engine +-- supplies index_id, as it does for GSIs. +-- +-- `search_schema` is nullable because the HASH element is optional (measured +-- against the live service): with one the search is partition-scoped and +-- SearchConditionExpression is required, without one it spans the table. +-- +-- `backfilling` mirrors the measured lifecycle: false while CREATING before the +-- scan starts, true while it runs, and the member is absent once ACTIVE. Stored +-- as an integer so the ACTIVE state is representable as NULL rather than as a +-- third boolean value. +CREATE TABLE IF NOT EXISTS vector_indexes ( + table_id TEXT NOT NULL, + index_id TEXT NOT NULL, + index_name TEXT NOT NULL, + dimensions INTEGER NOT NULL, + distance_function TEXT NOT NULL, + vector_attribute TEXT NOT NULL, + search_schema TEXT, + projection TEXT NOT NULL, + index_status TEXT NOT NULL DEFAULT 'CREATING', + backfilling INTEGER, + -- Items the backfill skipped because their stored bytes cannot enter the + -- index (unparseable row, malformed or wrong-dimension vector). NULL until + -- a backfill has completed; 0 afterwards when nothing was skipped. Kept so + -- an operator can see that an ACTIVE index deliberately omits rows, rather + -- than the build looping forever on them or dying part-way. + skipped_item_count INTEGER, + PRIMARY KEY (table_id, index_name), + CONSTRAINT vector_indexes_table_id_fkey + FOREIGN KEY (table_id) REFERENCES tables(table_id) ON DELETE CASCADE, + CONSTRAINT chk_vector_dimensions_positive CHECK (dimensions > 0), + CONSTRAINT chk_vector_backfilling_bool + CHECK (backfilling IS NULL OR backfilling IN (0, 1)), + -- An ACTIVE index must not carry the member at all, which is what the + -- service does. Enforced here as well as in core, so a bug in the backend + -- cannot persist a state the wire contract forbids. + CONSTRAINT chk_vector_active_has_no_backfilling + CHECK (index_status <> 'ACTIVE' OR backfilling IS NULL) +); + +CREATE UNIQUE INDEX IF NOT EXISTS idx_vector_indexes_index_id + ON vector_indexes (index_id); + +-- Resource tags. +CREATE TABLE IF NOT EXISTS tags ( + resource_arn TEXT NOT NULL, + tag_key TEXT NOT NULL, + tag_value TEXT NOT NULL, + PRIMARY KEY (resource_arn, tag_key) +); + +-- Migration tracking. +CREATE TABLE IF NOT EXISTS schema_history ( + filename TEXT PRIMARY KEY, + applied_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')) +); + +-- Settings (catalog version, data database name, runtime config). +CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL +); + +-- Stream shards. +CREATE TABLE IF NOT EXISTS stream_shards ( + shard_id TEXT PRIMARY KEY, + table_id TEXT NOT NULL REFERENCES tables(table_id) ON DELETE CASCADE, + parent_shard_id TEXT, + starting_sequence_number TEXT NOT NULL, + ending_sequence_number TEXT, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')) +); + +CREATE INDEX IF NOT EXISTS idx_stream_shards_table ON stream_shards (table_id); + +-- Stream records. +CREATE TABLE IF NOT EXISTS stream_records ( + shard_id TEXT NOT NULL REFERENCES stream_shards(shard_id) ON DELETE CASCADE, + sequence_number TEXT NOT NULL, + table_id TEXT NOT NULL, + event_name TEXT NOT NULL, + record_data TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + PRIMARY KEY (shard_id, sequence_number) +); + +CREATE INDEX IF NOT EXISTS idx_stream_records_created ON stream_records (created_at); + +-- Admin users. +CREATE TABLE IF NOT EXISTS admin_users ( + admin_name TEXT PRIMARY KEY, + password_hash TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')) +); + +-- IAM users. +CREATE TABLE IF NOT EXISTS iam_users ( + account_id TEXT NOT NULL REFERENCES accounts(account_id) ON DELETE CASCADE, + user_name TEXT NOT NULL, + user_arn TEXT NOT NULL UNIQUE, + password_hash TEXT, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + PRIMARY KEY (account_id, user_name) +); + +-- IAM user tags. +CREATE TABLE IF NOT EXISTS iam_user_tags ( + account_id TEXT NOT NULL, + user_name TEXT NOT NULL, + tag_key TEXT NOT NULL, + tag_value TEXT NOT NULL, + PRIMARY KEY (account_id, user_name, tag_key), + FOREIGN KEY (account_id, user_name) + REFERENCES iam_users(account_id, user_name) ON DELETE CASCADE +); + +-- Access keys. +CREATE TABLE IF NOT EXISTS access_keys ( + access_key_id TEXT PRIMARY KEY, + secret_key_encrypted BLOB NOT NULL, + account_id TEXT NOT NULL, + user_name TEXT NOT NULL, + is_active INTEGER NOT NULL DEFAULT 1, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + FOREIGN KEY (account_id, user_name) + REFERENCES iam_users(account_id, user_name) ON DELETE CASCADE +); + +-- IAM groups. +CREATE TABLE IF NOT EXISTS iam_groups ( + account_id TEXT NOT NULL REFERENCES accounts(account_id) ON DELETE CASCADE, + group_name TEXT NOT NULL, + group_arn TEXT NOT NULL UNIQUE, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + PRIMARY KEY (account_id, group_name) +); + +-- IAM group membership. +CREATE TABLE IF NOT EXISTS iam_group_members ( + account_id TEXT NOT NULL, + group_name TEXT NOT NULL, + user_name TEXT NOT NULL, + PRIMARY KEY (account_id, group_name, user_name), + FOREIGN KEY (account_id, group_name) + REFERENCES iam_groups(account_id, group_name) ON DELETE CASCADE, + FOREIGN KEY (account_id, user_name) + REFERENCES iam_users(account_id, user_name) ON DELETE CASCADE +); + +-- IAM roles. +CREATE TABLE IF NOT EXISTS iam_roles ( + account_id TEXT NOT NULL REFERENCES accounts(account_id) ON DELETE CASCADE, + role_name TEXT NOT NULL, + role_arn TEXT NOT NULL UNIQUE, + trust_policy TEXT NOT NULL, + permissions_boundary_arn TEXT, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + PRIMARY KEY (account_id, role_name) +); + +-- IAM role tags. +CREATE TABLE IF NOT EXISTS iam_role_tags ( + account_id TEXT NOT NULL, + role_name TEXT NOT NULL, + tag_key TEXT NOT NULL, + tag_value TEXT NOT NULL, + PRIMARY KEY (account_id, role_name, tag_key), + FOREIGN KEY (account_id, role_name) + REFERENCES iam_roles(account_id, role_name) ON DELETE CASCADE +); + +-- IAM sessions. +CREATE TABLE IF NOT EXISTS iam_sessions ( + session_token TEXT PRIMARY KEY, + access_key_id TEXT NOT NULL UNIQUE, + secret_key_encrypted BLOB NOT NULL, + account_id TEXT NOT NULL, + role_name TEXT NOT NULL, + session_name TEXT NOT NULL, + session_tags TEXT, + session_policy TEXT, + expires_at TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + FOREIGN KEY (account_id, role_name) + REFERENCES iam_roles(account_id, role_name) ON DELETE CASCADE +); + +-- IAM policies. +CREATE TABLE IF NOT EXISTS iam_policies ( + account_id TEXT NOT NULL REFERENCES accounts(account_id) ON DELETE CASCADE, + principal_type TEXT NOT NULL CHECK (principal_type IN ('user', 'group', 'role')), + principal_name TEXT NOT NULL, + policy_name TEXT NOT NULL, + policy_document TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + PRIMARY KEY (account_id, principal_type, principal_name, policy_name) +); + +-- IAM permissions boundaries. +CREATE TABLE IF NOT EXISTS iam_permissions_boundaries ( + account_id TEXT NOT NULL REFERENCES accounts(account_id) ON DELETE CASCADE, + principal_type TEXT NOT NULL CHECK (principal_type IN ('user', 'role')), + principal_name TEXT NOT NULL, + policy_document TEXT NOT NULL, + PRIMARY KEY (account_id, principal_type, principal_name) +); + +-- Idempotency tokens for TransactWriteItems. +CREATE TABLE IF NOT EXISTS idempotency_tokens ( + account_id TEXT NOT NULL, + token TEXT NOT NULL, + fingerprint TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + PRIMARY KEY (account_id, token) +); + +CREATE INDEX IF NOT EXISTS idx_idempotency_tokens_created ON idempotency_tokens (created_at); + +-- Metrics (1-minute aggregation). ±9e999 store as ±Infinity in SQLite REAL, +-- matching PostgreSQL's Infinity / -Infinity seed for min / max. +CREATE TABLE IF NOT EXISTS metrics ( + bucket TEXT NOT NULL, + metric TEXT NOT NULL, + table_name TEXT NOT NULL DEFAULT '', + index_name TEXT NOT NULL DEFAULT '', + operation TEXT NOT NULL DEFAULT '', + sum REAL NOT NULL DEFAULT 0, + count INTEGER NOT NULL DEFAULT 0, + min REAL NOT NULL DEFAULT 9e999, + max REAL NOT NULL DEFAULT -9e999, + PRIMARY KEY (bucket, metric, table_name, index_name, operation) +); + +CREATE INDEX IF NOT EXISTS idx_metrics_bucket ON metrics (bucket); + +-- Login attempt tracking. +CREATE TABLE IF NOT EXISTS login_attempts ( + principal TEXT NOT NULL, + attempted_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')), + success INTEGER NOT NULL, + source_ip TEXT +); + +CREATE INDEX IF NOT EXISTS idx_login_attempts_principal_time + ON login_attempts (principal, attempted_at DESC); + +CREATE INDEX IF NOT EXISTS idx_login_attempts_source_ip_time + ON login_attempts (source_ip, attempted_at DESC) + WHERE source_ip IS NOT NULL; + +-- Backup metadata. +CREATE TABLE IF NOT EXISTS backups ( + backup_arn TEXT PRIMARY KEY, + backup_name TEXT NOT NULL, + table_id TEXT NOT NULL, + table_name TEXT NOT NULL, + account_id TEXT NOT NULL, + backup_status TEXT NOT NULL DEFAULT 'AVAILABLE', + backup_type TEXT NOT NULL DEFAULT 'USER', + backup_size_bytes INTEGER NOT NULL DEFAULT 0, + item_count INTEGER NOT NULL DEFAULT 0, + key_schema TEXT NOT NULL, + attribute_definitions TEXT NOT NULL, + billing_mode TEXT NOT NULL DEFAULT 'PAY_PER_REQUEST', + provisioned_throughput TEXT, + stream_specification TEXT, + created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')) +); + +CREATE INDEX IF NOT EXISTS idx_backups_table ON backups (account_id, table_name); + +-- Backup items. +CREATE TABLE IF NOT EXISTS backup_items ( + backup_arn TEXT NOT NULL REFERENCES backups(backup_arn) ON DELETE CASCADE, + pk TEXT NOT NULL, + sk TEXT, + item_data TEXT NOT NULL +); + +CREATE INDEX IF NOT EXISTS idx_backup_items_arn ON backup_items (backup_arn); + +-- Continuous backups / PITR status. +CREATE TABLE IF NOT EXISTS continuous_backups ( + account_id TEXT NOT NULL, + table_name TEXT NOT NULL, + pitr_enabled INTEGER NOT NULL DEFAULT 0, + earliest_restorable TEXT, + latest_restorable TEXT, + PRIMARY KEY (account_id, table_name) +); + +-- Persistent queue for async GSI propagation. A row is inserted inside the +-- base write transaction (zero crash window) and consumed by a background +-- worker once `ready_at` has passed; survives process crash/restart. +-- +-- One row per async index: each row is self-describing — `index_context` +-- carries the base key schema, attribute definitions, and the single target +-- index definition captured at enqueue, so the worker applies with zero +-- catalog reads. `worker_partition` is a stable hash of the base table key; +-- all updates to a given base item share a partition and `ready_at` is kept +-- monotonically non-decreasing within it, so the worker (which drains in `id` +-- order) preserves per-key FIFO even with randomized propagation jitter. +CREATE TABLE IF NOT EXISTS gsi_pending ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + table_id TEXT NOT NULL, + worker_partition INTEGER NOT NULL, + old_item TEXT, + new_item TEXT, + index_context TEXT NOT NULL, + ready_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')) +); + +CREATE INDEX IF NOT EXISTS idx_gsi_pending_claim + ON gsi_pending (worker_partition, ready_at, id); + +-- Monotonic counters. Replaces PostgreSQL sequences. The stream counter is +-- seeded to the current epoch in microseconds so sequence numbers are +-- time-ordered and survive restarts (mirrors the PostgreSQL setval seed). +CREATE TABLE IF NOT EXISTS seq_counters ( + name TEXT PRIMARY KEY, + value INTEGER NOT NULL DEFAULT 0 +); + +INSERT OR IGNORE INTO seq_counters (name, value) + VALUES ('stream', CAST(strftime('%s','now') AS INTEGER) * 1000000); + +-- Seed settings (mirror PostgreSQL defaults). +-- Recorded catalog version. Upserts rather than INSERT OR IGNORE: this schema is +-- re-applied by `extenddb migrate`, and with IGNORE the recorded version would +-- never advance, so a migration could add objects and still leave the server +-- refusing to start on a version mismatch. Must stay in step with +-- `CATALOG_VERSION` above; they are checked against each other in a test. +INSERT INTO settings (key, value) VALUES ('catalog_version', '0.0.3') + ON CONFLICT(key) DO UPDATE SET value = excluded.value; +INSERT OR IGNORE INTO settings (key, value) VALUES ('control_plane_delay_seconds', '0.25'); +INSERT OR IGNORE INTO settings (key, value) VALUES ('index_propagation_delay_ms', '10'); diff --git a/crates/storage/src/backup_definition.rs b/crates/storage/src/backup_definition.rs new file mode 100644 index 000000000..8f0c496cc --- /dev/null +++ b/crates/storage/src/backup_definition.rs @@ -0,0 +1,552 @@ +// Copyright 2026 ExtendDB contributors +// SPDX-License-Identifier: Apache-2.0 + +//! The table definition a backup records, shared by the SQL backends. +//! +//! RestoreTableFromBackup reproduces what the service reproduces: key schema, +//! attribute definitions, global and local secondary indexes, billing mode and +//! provisioned throughput, table class, and encryption settings. Streams, TTL, +//! tags, deletion protection, and point-in-time recovery settings are not +//! restored, matching the service. The key schema and attribute definitions +//! keep their own catalog columns; everything else lives here. +//! +//! Stored in the wire's own shape behind a version marker, not as a copy of +//! catalog rows. A backup outlives the schema that produced it, so freezing +//! physical column names into it would let a later catalog change silently +//! alter the meaning of backups already on disk. A reader that finds a version +//! it does not know refuses the restore rather than guessing. + +use extenddb_core::types::{ + BillingMode, CreateTableInput, GsiInput, KeySchemaElement, KeyType, LsiInput, + OnDemandThroughput, ProvisionedThroughput, +}; +use serde::{Deserialize, Serialize}; + +use crate::error::StorageError; + +/// Items a backup or a restore buffers before writing them out, at most. +pub const COPY_BATCH_ITEMS: usize = 500; + +/// Bytes of stored item JSON a backup or a restore buffers before writing +/// them out, at most (one item over the budget is still taken whole). Counted +/// on the stored text, not the DynamoDB item size, because the text is what +/// is held: a 400 KB item can be several MB of JSON (a list of booleans is +/// about seven times its DynamoDB size). +pub const COPY_BATCH_BYTES: usize = 4 * 1024 * 1024; + +/// Capacity overrides for one `RestoreTableFromBackup` request. +#[derive(Debug, Clone, Default, PartialEq)] +pub struct RestoreTableOverrides { + /// Replacement billing mode, if the request supplied one. + pub billing_mode: Option, + /// Replacement table throughput, if the request supplied one. + pub provisioned_throughput: Option, +} + +impl RestoreTableOverrides { + /// Apply overrides to a legacy backup's populated `CreateTableInput`. + /// + /// Legacy backups do not contain a [`BackupTableDefinition`], so backends + /// first reconstruct the source defaults and then call this method. + pub fn apply_to_create_input(&self, input: &mut CreateTableInput) { + if let Some(billing_mode) = self.billing_mode { + input.billing_mode = Some(billing_mode); + } + if let Some(throughput) = &self.provisioned_throughput { + input.provisioned_throughput = Some(throughput.clone()); + } + + // Apply overrides before normalizing throughput for on-demand billing. + if matches!( + input.billing_mode.as_ref(), + Some(BillingMode::PayPerRequest) + ) { + input.provisioned_throughput = None; + for index in input.global_secondary_indexes.iter_mut().flatten() { + index.provisioned_throughput = None; + } + } + } +} + +/// Refuse to restore a table whose base key has more than one HASH or more +/// than one RANGE attribute. +/// +/// Multi-part base keys are an opt-in preview (`enable_multipart_keys`), and +/// the item read and write paths of the SQL backends address only the first +/// HASH and first RANGE attribute of a base table. A restore that laid such a +/// table out by its full key would produce rows those paths cannot find; one +/// that followed the read paths would collapse distinct items. Refusing is the +/// only result that is not silently wrong. +/// +/// # Errors +/// +/// [`StorageError::Unsupported`] for a multi-part base key. +pub fn ensure_single_part_base_key( + key_schema: &[KeySchemaElement], + backup_arn: &str, +) -> Result<(), StorageError> { + let hashes = key_schema + .iter() + .filter(|k| k.key_type == KeyType::Hash) + .count(); + let ranges = key_schema + .iter() + .filter(|k| k.key_type == KeyType::Range) + .count(); + if hashes <= 1 && ranges <= 1 { + return Ok(()); + } + Err(StorageError::Unsupported(format!( + "backup {backup_arn} is of a table with a multi-part key ({hashes} HASH, {ranges} \ + RANGE attributes); restoring it is not supported" + ))) +} + +/// The only definition version this build writes and reads. +pub const BACKUP_DEFINITION_VERSION: u32 = 1; + +/// The parts of a table's definition a restore recreates, beyond its keys. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct BackupTableDefinition { + #[serde(rename = "Version")] + pub version: u32, + /// `PROVISIONED` or `PAY_PER_REQUEST`. + #[serde(rename = "BillingMode")] + pub billing_mode: String, + #[serde( + rename = "ProvisionedThroughput", + default, + skip_serializing_if = "Option::is_none" + )] + pub provisioned_throughput: Option, + #[serde(rename = "GlobalSecondaryIndexes", default)] + pub global_secondary_indexes: Vec, + #[serde(rename = "LocalSecondaryIndexes", default)] + pub local_secondary_indexes: Vec, + #[serde( + rename = "TableClass", + default, + skip_serializing_if = "Option::is_none" + )] + pub table_class: Option, + #[serde( + rename = "SSESpecification", + default, + skip_serializing_if = "Option::is_none" + )] + pub sse_specification: Option, + #[serde( + rename = "OnDemandThroughput", + default, + skip_serializing_if = "Option::is_none" + )] + pub on_demand_throughput: Option, + /// Names of the source table's vector indexes. Restore refuses a backup + /// that has any, rather than producing a table without them. + #[serde(rename = "VectorIndexNames", default)] + pub vector_index_names: Vec, +} + +impl BackupTableDefinition { + /// Parse a stored definition, refusing a version this build cannot read. + /// + /// # Errors + /// + /// [`StorageError::Unsupported`] for an unknown version, and + /// [`StorageError::Internal`] for a document that does not parse. + pub fn from_json(value: serde_json::Value, backup_arn: &str) -> Result { + let version = value.get("Version").and_then(serde_json::Value::as_u64); + if version != Some(u64::from(BACKUP_DEFINITION_VERSION)) { + return Err(StorageError::Unsupported(format!( + "backup {backup_arn} carries a table definition this build cannot read \ + (version {version:?})" + ))); + } + serde_json::from_value(value).map_err(|e| { + StorageError::Internal(format!("backup {backup_arn} table definition: {e}")) + }) + } + + /// Serialize for storage. + /// + /// # Errors + /// + /// [`StorageError::Internal`] if serialization fails. + pub fn to_json(&self) -> Result { + let mut definition = self.clone(); + definition.normalize_provisioned_throughput(); + serde_json::to_value(definition).map_err(|e| StorageError::Internal(e.to_string())) + } + + /// Remove catalog throughput left behind by a switch to on-demand billing. + fn normalize_provisioned_throughput(&mut self) { + if self.billing_mode != "PAY_PER_REQUEST" { + return; + } + self.provisioned_throughput = None; + for index in &mut self.global_secondary_indexes { + index.provisioned_throughput = None; + } + } + + /// Refuse a definition the restore cannot reproduce in full. + /// + /// # Errors + /// + /// [`StorageError::Unsupported`] if the source table had vector indexes. + pub fn ensure_restorable(&self, backup_arn: &str) -> Result<(), StorageError> { + if self.vector_index_names.is_empty() { + return Ok(()); + } + Err(StorageError::Unsupported(format!( + "backup {backup_arn} has {} vector index(es); restoring a table with vector \ + indexes is not supported by this storage backend", + self.vector_index_names.len() + ))) + } + + /// Apply the definition and explicit request overrides to a restore target's + /// `CreateTableInput`. + /// + /// # Errors + /// + /// [`StorageError::Validation`] when switching a backup with a GSI that has + /// no provisioned throughput to provisioned billing. The request would need + /// `GlobalSecondaryIndexOverride`, which ExtendDB does not support. + pub fn apply_to( + mut self, + input: &mut CreateTableInput, + overrides: &RestoreTableOverrides, + ) -> Result<(), StorageError> { + if let Some(billing_mode) = overrides.billing_mode { + self.billing_mode = match billing_mode { + BillingMode::Provisioned => "PROVISIONED".to_owned(), + BillingMode::PayPerRequest => "PAY_PER_REQUEST".to_owned(), + }; + } + if let Some(throughput) = &overrides.provisioned_throughput { + self.provisioned_throughput = Some(throughput.clone()); + } + + if self.billing_mode == "PROVISIONED" + && (overrides.billing_mode.is_some() || overrides.provisioned_throughput.is_some()) + && let Some(index) = self.global_secondary_indexes.iter().find(|index| { + index + .provisioned_throughput + .as_ref() + .is_none_or(|throughput| { + throughput.read_capacity_units < 1 || throughput.write_capacity_units < 1 + }) + }) + { + return Err(StorageError::Validation(format!( + "One or more parameter values were invalid: GlobalSecondaryIndexOverride must \ + be specified for index: {} when BillingModeOverride is PROVISIONED", + index.index_name + ))); + } + + // Apply overrides before normalizing throughput for on-demand billing. + self.normalize_provisioned_throughput(); + let on_demand = self.billing_mode == "PAY_PER_REQUEST"; + input.billing_mode = Some(if on_demand { + BillingMode::PayPerRequest + } else { + BillingMode::Provisioned + }); + input.provisioned_throughput = self.provisioned_throughput; + input.global_secondary_indexes = + (!self.global_secondary_indexes.is_empty()).then_some(self.global_secondary_indexes); + input.local_secondary_indexes = + (!self.local_secondary_indexes.is_empty()).then_some(self.local_secondary_indexes); + input.table_class = self.table_class; + input.sse_specification = self.sse_specification; + input.on_demand_throughput = self.on_demand_throughput; + Ok(()) + } +} + +/// Throughput stored in a catalog `provisioned_throughput` column, which holds +/// either the request shape or the description shape depending on the writer. +/// Both carry the two capacity members under the same names. +#[must_use] +pub fn throughput_from_catalog(value: Option<&serde_json::Value>) -> Option { + let v = value?; + let read = v.get("ReadCapacityUnits")?.as_i64()?; + let write = v.get("WriteCapacityUnits")?.as_i64()?; + Some(ProvisionedThroughput { + read_capacity_units: read, + write_capacity_units: write, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use extenddb_core::types::{KeySchemaElement, KeyType, Projection, ProjectionType}; + + fn sample() -> BackupTableDefinition { + BackupTableDefinition { + version: BACKUP_DEFINITION_VERSION, + billing_mode: "PROVISIONED".to_owned(), + provisioned_throughput: Some(ProvisionedThroughput { + read_capacity_units: 7, + write_capacity_units: 9, + }), + global_secondary_indexes: vec![GsiInput { + index_name: "g".to_owned(), + key_schema: vec![KeySchemaElement { + attribute_name: "gpk".to_owned(), + key_type: KeyType::Hash, + }], + projection: Projection { + projection_type: ProjectionType::KeysOnly, + non_key_attributes: None, + }, + provisioned_throughput: Some(ProvisionedThroughput { + read_capacity_units: 3, + write_capacity_units: 4, + }), + }], + local_secondary_indexes: Vec::new(), + table_class: Some("STANDARD_INFREQUENT_ACCESS".to_owned()), + sse_specification: None, + on_demand_throughput: None, + vector_index_names: Vec::new(), + } + } + + #[test] + fn refuses_multi_part_base_keys() { + let k = |n: &str, t| KeySchemaElement { + attribute_name: n.to_owned(), + key_type: t, + }; + assert!(ensure_single_part_base_key(&[k("a", KeyType::Hash)], "arn").is_ok()); + assert!( + ensure_single_part_base_key(&[k("a", KeyType::Hash), k("b", KeyType::Range)], "arn") + .is_ok() + ); + for ks in [ + vec![k("a", KeyType::Hash), k("b", KeyType::Hash)], + vec![ + k("a", KeyType::Hash), + k("b", KeyType::Range), + k("c", KeyType::Range), + ], + ] { + assert!(matches!( + ensure_single_part_base_key(&ks, "arn"), + Err(StorageError::Unsupported(_)) + )); + } + } + + #[test] + fn round_trips_through_json() { + let def = sample(); + let back = BackupTableDefinition::from_json(def.to_json().unwrap(), "arn").unwrap(); + assert_eq!(back, def); + } + + #[test] + fn refuses_an_unknown_version() { + let mut json = sample().to_json().unwrap(); + json["Version"] = serde_json::json!(2); + let err = BackupTableDefinition::from_json(json, "arn").unwrap_err(); + assert!(matches!(err, StorageError::Unsupported(_)), "{err:?}"); + } + + #[test] + fn refuses_vector_indexes() { + let mut def = sample(); + def.vector_index_names.push("v".to_owned()); + assert!(matches!( + def.ensure_restorable("arn"), + Err(StorageError::Unsupported(_)) + )); + } + + #[test] + fn applies_to_a_create_input() { + let mut input = CreateTableInput::default(); + sample() + .apply_to(&mut input, &RestoreTableOverrides::default()) + .unwrap(); + assert_eq!(input.billing_mode, Some(BillingMode::Provisioned)); + assert_eq!( + input + .provisioned_throughput + .as_ref() + .map(|p| p.read_capacity_units), + Some(7) + ); + assert_eq!( + input.global_secondary_indexes.as_ref().map(Vec::len), + Some(1) + ); + assert!(input.local_secondary_indexes.is_none()); + assert_eq!( + input.table_class.as_deref(), + Some("STANDARD_INFREQUENT_ACCESS") + ); + } + + #[test] + fn provisioned_table_round_trip_keeps_throughput() { + let definition = sample(); + let captured = definition.to_json().unwrap(); + let restored = BackupTableDefinition::from_json(captured, "arn").unwrap(); + let mut input = CreateTableInput::default(); + restored + .apply_to(&mut input, &RestoreTableOverrides::default()) + .unwrap(); + + assert_eq!(input.billing_mode, Some(BillingMode::Provisioned)); + assert_eq!( + input.provisioned_throughput.map(|throughput| ( + throughput.read_capacity_units, + throughput.write_capacity_units + )), + Some((7, 9)) + ); + assert_eq!( + input.global_secondary_indexes.unwrap()[0] + .provisioned_throughput + .as_ref() + .map(|throughput| ( + throughput.read_capacity_units, + throughput.write_capacity_units + )), + Some((3, 4)) + ); + } + + #[test] + fn pay_per_request_table_with_stale_throughput_is_normalized() { + let mut definition = sample(); + definition.billing_mode = "PAY_PER_REQUEST".to_owned(); + definition.global_secondary_indexes.clear(); + + let captured = definition.to_json().unwrap(); + assert!(captured.get("ProvisionedThroughput").is_none()); + + let mut input = CreateTableInput::default(); + definition + .apply_to(&mut input, &RestoreTableOverrides::default()) + .unwrap(); + assert_eq!(input.billing_mode, Some(BillingMode::PayPerRequest)); + assert!(input.provisioned_throughput.is_none()); + } + + #[test] + fn pay_per_request_gsi_with_stale_throughput_is_normalized() { + let mut definition = sample(); + definition.billing_mode = "PAY_PER_REQUEST".to_owned(); + definition.provisioned_throughput = None; + + let captured = definition.to_json().unwrap(); + assert!( + captured["GlobalSecondaryIndexes"][0] + .get("ProvisionedThroughput") + .is_none() + ); + + let mut input = CreateTableInput::default(); + definition + .apply_to(&mut input, &RestoreTableOverrides::default()) + .unwrap(); + assert!( + input.global_secondary_indexes.unwrap()[0] + .provisioned_throughput + .is_none() + ); + } + + #[test] + fn reads_both_catalog_throughput_shapes() { + let input = serde_json::json!({"ReadCapacityUnits": 5, "WriteCapacityUnits": 6}); + let desc = serde_json::json!({ + "ReadCapacityUnits": 5, "WriteCapacityUnits": 6, "NumberOfDecreasesToday": 0 + }); + for v in [input, desc] { + let pt = throughput_from_catalog(Some(&v)).unwrap(); + assert_eq!((pt.read_capacity_units, pt.write_capacity_units), (5, 6)); + } + assert!(throughput_from_catalog(None).is_none()); + } + + #[test] + fn pay_per_request_override_drops_table_and_gsi_throughput() { + let overrides = RestoreTableOverrides { + billing_mode: Some(BillingMode::PayPerRequest), + provisioned_throughput: None, + }; + let mut input = CreateTableInput::default(); + sample().apply_to(&mut input, &overrides).unwrap(); + + assert_eq!(input.billing_mode, Some(BillingMode::PayPerRequest)); + assert!(input.provisioned_throughput.is_none()); + assert!( + input.global_secondary_indexes.unwrap()[0] + .provisioned_throughput + .is_none() + ); + } + + #[test] + fn provisioned_override_with_gsi_without_throughput_is_refused_before_apply() { + let mut definition = sample(); + definition.billing_mode = "PAY_PER_REQUEST".to_owned(); + definition.provisioned_throughput = None; + definition.global_secondary_indexes[0].provisioned_throughput = None; + let overrides = RestoreTableOverrides { + billing_mode: Some(BillingMode::Provisioned), + provisioned_throughput: Some(ProvisionedThroughput { + read_capacity_units: 5, + write_capacity_units: 5, + }), + }; + let mut input = CreateTableInput::default(); + + let err = definition.apply_to(&mut input, &overrides).unwrap_err(); + + match err { + StorageError::Validation(message) => assert_eq!( + message, + "One or more parameter values were invalid: GlobalSecondaryIndexOverride must \ + be specified for index: g when BillingModeOverride is PROVISIONED" + ), + other => panic!("expected Validation error, got {other:?}"), + } + assert_eq!(input, CreateTableInput::default()); + } + + #[test] + fn provisioned_override_replaces_table_throughput() { + let mut definition = sample(); + definition.billing_mode = "PAY_PER_REQUEST".to_owned(); + definition.provisioned_throughput = None; + definition.global_secondary_indexes.clear(); + + let overrides = RestoreTableOverrides { + billing_mode: Some(BillingMode::Provisioned), + provisioned_throughput: Some(ProvisionedThroughput { + read_capacity_units: 5, + write_capacity_units: 5, + }), + }; + let mut input = CreateTableInput::default(); + definition.apply_to(&mut input, &overrides).unwrap(); + + assert_eq!(input.billing_mode, Some(BillingMode::Provisioned)); + assert_eq!( + input.provisioned_throughput.map(|throughput| ( + throughput.read_capacity_units, + throughput.write_capacity_units + )), + Some((5, 5)) + ); + } +} diff --git a/crates/storage/src/error.rs b/crates/storage/src/error.rs index b98593c09..563dfcba6 100755 --- a/crates/storage/src/error.rs +++ b/crates/storage/src/error.rs @@ -51,6 +51,10 @@ pub enum StorageError { /// measured 2026-08-13). #[error("{0}")] LimitExceeded(String), + /// A backup an in-progress restore is still reading. Maps to + /// `BackupInUseException`. + #[error("{0}")] + BackupInUse(String), /// A failure that is expected to succeed on retry: I/O errors, pool /// timeouts, SQLITE_BUSY / SQLITE_LOCKED. Exists so queue workers can tell /// "this row can never be applied" (drop it, or the whole queue stalls) diff --git a/crates/storage/src/lib.rs b/crates/storage/src/lib.rs index 82593548d..50cf3546d 100755 --- a/crates/storage/src/lib.rs +++ b/crates/storage/src/lib.rs @@ -9,6 +9,7 @@ pub mod authorization_store; pub mod backend; +pub mod backup_definition; pub mod bootstrapper; pub mod config; pub mod diagnostics; @@ -24,6 +25,7 @@ pub mod vector_catalog; pub mod vector_lifecycle; pub use backend::{Backend, BackendAlreadySet, backend_name, set_backend, try_backend}; +pub use backup_definition::RestoreTableOverrides; pub use transact::{IdempotencyKey, TransactGetOp, TransactWriteOp}; @@ -700,12 +702,14 @@ pub trait BackupEngine: Send + Sync { /// /// `account_id` is the caller's account: it owns the new table *and* scopes /// the source backup lookup, since a backup can only be restored by the - /// account that owns it. + /// account that owns it. `overrides` explicitly supplies any validated + /// billing mode and provisioned throughput replacements for the new table. fn restore_table_from_backup( &self, account_id: &str, target_table_name: &str, backup_arn: &str, + overrides: RestoreTableOverrides, ) -> BoxFuture<'_, Result>; /// Describe continuous backups / PITR status for a table. diff --git a/docs/differences-from-dynamodb.md b/docs/differences-from-dynamodb.md index 447c54c8e..2460ae993 100755 --- a/docs/differences-from-dynamodb.md +++ b/docs/differences-from-dynamodb.md @@ -70,12 +70,28 @@ adaptation when switching between ExtendDB and the real service. | Vector index update propagation | Eventually consistent, the same model as a GSI | Matches DynamoDB. Maintenance is queued on the same propagation queue as async GSIs, so a search immediately after a write may not see it. Governed by the same `index_propagation_delay_ms` setting; unlike a GSI there is no per-index override. A value of 0 applies maintenance inline in the write's own transaction, which is stricter than the service and exists so a test can assert steady state without waiting. Inline additionally requires the table's propagation queue to be empty: while rows queued during an index build are still draining, a new write queues behind them instead of overtaking them, so a brief search-after-write window exists even at 0 until the queue drains. That window is ordering, not data loss; the write is applied, in order, by the queue worker. That zero behaves differently while an index is still building, and the two backends differ: **PostgreSQL** applies inline only to an index that is already `ACTIVE`, and defers a write to a building index whatever the delay says, because the write must not reach the index ahead of the backfill's older snapshot of the same item; **SQLite** does not check the status on this path, so with a zero delay a write during a backfill is applied inline and bypasses the hold that keeps the two ordered. The SQLite behaviour is tracked as a defect (F-20), not intended, and it is reachable only with the zero delay, which is a test setting. | | `SearchVectors` score at extreme magnitudes (PostgreSQL backend only) | Returns the true distance as a number for any vector of finite components | The score is bounded to a finite value instead, and the bound is not a measured service answer. pgvector accumulates distances in single precision, so magnitudes far below `f32::MAX` overflow inside the extension: Euclidean above about 9.2e18, dot product above about 1.8e19, and cosine at both ends, above about 1.8e19 and below about 3.7e-23, which is where a component's single-precision square rounds to zero. A non-finite score cannot be serialised as JSON at all, so the result is bounded in SQL: `1e308` for an overflowed distance, `-1e308` for an overflowed negated inner product, which the score contract negates so a client sees `1e308` in `Score`, and 1.0 for a cosine that comes back NaN. Ranking is unaffected at the overflow end, because each bound sits at the end its metric overflows towards, so the farthest row stays farthest and the most similar stays most similar; two rows that both overflow tie, and the tie breaks on the base key. At the underflow end cosine loses resolution rather than being bounded, and which side is tiny decides how much. For a tiny **query** vector, its norm underflows while the inner product usually does not, so the quotient is an infinity that pgvector clamps and the reported distance collapses to one of 0, 1 or 2 following the sign of the inner product, with the 1.0 substitute firing only when the vectors are exactly orthogonal and the quotient is therefore 0/0 (measured). For a tiny **stored** vector it is worse: the stored norm is computed in single precision and reaches zero, so the zero-vector guard fires and that row reports 1.0 at every angle, parallel included. A corpus of tiny embeddings therefore loses ranking altogether rather than losing resolution. For a tiny query vector, by contrast, ranking still separates nearer-than-orthogonal from farther and loses resolution only within each half. The SQLite backend owns its own arithmetic, computes in double precision, and reports the true value, which is why this row is scoped to PostgreSQL. | | Vector index deletion window | `UpdateTable` Delete leaves the index in `DELETING` long enough to observe, then removes it | No observable `DELETING` window on either backend: the catalog row is removed inside the `UpdateTable` transaction, so a `DescribeTable` immediately afterwards already omits the index. The index's data table is dropped after that commit, in a separate transaction, and the two backends handle a failure there differently. **PostgreSQL** treats it as best effort, because the data table lives in a different database entirely: a failure is logged and skipped rather than failing the request. **SQLite** propagates it, so a failed drop returns an error from an `UpdateTable` whose catalog change has already committed, which is the more surprising outcome of the two: the index is gone from the catalog and the caller saw a failure. So an operator debugging a leftover `_ddb_vec_*` table should look for that warning rather than assume the delete was incomplete. | -| Restoring a backup of a table that had vector indexes | Restores the table with its vector indexes intact: the configuration survives, items keep their vector attributes, and `SearchVectors` works as soon as the table is `ACTIVE` (measured) | Neither backend restores the indexes, and the two fail differently. **PostgreSQL refuses the restore** with a `ValidationException` naming the backup and the index count, because restore does not carry index data across and a table that looks restored while answering every search with nothing is worse than a refusal a caller can act on. **SQLite does not refuse**: its backup path does not capture vector indexes at all, so a restore silently produces the table without them. That silence is tracked as a defect rather than intended, and it is the reason the PostgreSQL path refuses instead of matching it. A backup taken from a table with no vector indexes restores normally on both. | +| Restoring a backup of a table that had vector indexes | Restores the table with its vector indexes intact: the configuration survives, items keep their vector attributes, and `SearchVectors` works as soon as the table is `ACTIVE` (measured) | Neither backend restores the indexes, and both refuse the restore with a `ValidationException` naming the backup and the index count, because restore does not carry index data across and a table that looks restored while answering every search with nothing is worse than a refusal a caller can act on. A backup taken from a table with no vector indexes restores normally on both. | | `SearchVectors` endpoint | Served only on `search-dynamodb..amazonaws.com`. The standard `dynamodb..amazonaws.com` endpoint answers the same request with HTTP 400 `UnknownOperationException` ("This operation is not supported by this endpoint"); every control-plane and item operation stays on the standard endpoint. Signing is unchanged either way (service name `dynamodb`, target prefix `DynamoDB_20120810`) | Served on the same endpoint as every other operation, so a client is pointed at one ExtendDB endpoint for all of them. Two consequences worth knowing: an SDK that resolves a separate search hostname from its endpoint ruleset needs its endpoint overridden to reach ExtendDB, and ExtendDB does **not** reproduce the service's refusal, so a test asserting `UnknownOperationException` for vector search on the base endpoint passes against Amazon DynamoDB and fails here. | | `SearchVectors` result order for equal distances | Measured unstable: three identical searches returned tied rows in three different orders, and a top-k that truncates a tie group keeps an arbitrary subset of it | Deterministic on both backends, by different means. **PostgreSQL** sorts explicitly on the base table's full primary key after the score, so the order is a property of the query. **SQLite** issues no `ORDER BY` and resolves ties by scan order through a stable top-k, so its order is a property of the plan rather than something the query guarantees. Either way a client re-issuing an identical search sees the same order, which is stricter than the service rather than divergent in outcome. Do not rely on the two backends agreeing on which subset of a truncated tie group they keep. | | Vector indexes on the MongoDB backend | Vector indexes and `SearchVectors` are available | Not supported at all. That backend provides no vector search implementation, so the engine's capability gate refuses `CreateTable` and `UpdateTable` carrying `VectorIndexes` and every `SearchVectors` request, with the same capability message a PostgreSQL deployment without pgvector returns. Every other vector row in this document describes the PostgreSQL and SQLite backends. | | Multi-part base table keys | Not supported | Preview extension (opt-in via `enable_multipart_keys` setting). Standard single/composite keys work identically. | +## Backup and Restore + +| Area | DynamoDB | ExtendDB | +|------|----------|------| +| What a restore carries across | Key schema, attribute definitions, items, secondary indexes, billing mode and provisioned throughput, SSE settings. Stream settings, TTL, tags, auto scaling, IAM policies, and CloudWatch settings must be reapplied by hand | The same set on PostgreSQL, SQLite, and MongoDB, plus table class and on-demand throughput. Streams, TTL, tags, and deletion protection are not restored. A backup written by a version before catalog 0.0.4 carries no table definition and restores as keys and items only, with 5/5 provisioned throughput | +| `RestoreTableFromBackup` overrides | Current request members are `BillingModeOverride`, `ProvisionedThroughputOverride`, `GlobalSecondaryIndexOverride`, `LocalSecondaryIndexOverride`, `SSESpecificationOverride`, `OnDemandThroughputOverride`, and `VectorIndexOverride`; DynamoDB applies them to the restored table | PostgreSQL, SQLite, and MongoDB support `BillingModeOverride` and `ProvisionedThroughputOverride`. Switching to `PAY_PER_REQUEST` drops table and GSI throughput. `GlobalSecondaryIndexOverride`, `LocalSecondaryIndexOverride`, `SSESpecificationOverride`, `OnDemandThroughputOverride`, and `VectorIndexOverride` are refused with `ValidationException` before the target table is created. Switching an on-demand backup with GSIs to `PROVISIONED` is likewise refused because it requires `GlobalSecondaryIndexOverride` | +| Multi-part base table keys (preview) | Not supported | A backup of a table created with `enable_multipart_keys` is refused at restore with `ValidationException`; the item paths address such tables by their first HASH and RANGE attribute only | +| `DeleteBackup` during a restore from that backup | `BackupInUseException` | `BackupInUseException` (HTTP 400) until the restore completes or fails, on all three backends. PostgreSQL and SQLite register the restore in the same transaction that creates its target, with the backup row held so a concurrent `DeleteBackup` orders strictly before (the restore then sees no backup) or after (it sees the restore). MongoDB has no cross-collection transaction and uses a claim marker on the backup's metadata instead: a restore claims the backup before reading it and releases the claim once its `CREATING` target exists; `DeleteBackup` claims it the same way before dropping data. A claim left by a crashed process is cleared at the next startup, so a backup is never stuck undeletable; until then a restore or delete of that backup gets `BackupInUseException` | +| `DeleteTable` on a table being restored | `ResourceInUseException` | PostgreSQL, SQLite, and MongoDB match on a restore target. On all three backends an ordinary table in `CREATING` can still be deleted during its control-plane delay, where the service refuses | +| `RestoreSummary` lifecycle | The immediate restore response reports `RestoreInProgress=true`; later `DescribeTable` calls report `false` after completion and retain the same `RestoreDateTime` | PostgreSQL, SQLite, and MongoDB do the same. The engine deliberately forces `RestoreInProgress=true` on the immediate response even when a synchronous backend has already finished copying; the backend's recorded `RestoreDateTime` is reused by later `DescribeTable` calls | +| Restore interrupted by a server crash | Managed by the service | PostgreSQL removes an abandoned `CREATING` restore target on a control-plane pass; SQLite removes one when the next server starts. MongoDB has no abandoned-restore sweep: the target remains `CREATING`, the API continues to refuse `DeleteTable`, and an operator must remove its catalog and data collections by hand | +| Point-in-time recovery | 35-day window | Not supported; `UpdateContinuousBackups` and `RestoreTableToPointInTime` are refused | +| Where backups live | Managed storage, independent of the table | Inside the database the server already uses, so a lost database loses its backups too: on PostgreSQL in the catalog database (`backups`, `backup_items`, `backup_definitions`, `table_restores`), on SQLite in the one database file, on MongoDB as a `backups` metadata collection in the catalog database and one data collection per backup in the data database. Take database-level backups for off-host copies until the backup store program ships | +| Backup creation status | `CreateBackup` returns while the backup is `CREATING`; `ListBackups` and `DescribeBackup` expose that state until it becomes `AVAILABLE` | SQLite similarly exposes `CREATING` while it copies: size and item count remain zero, `DeleteBackup` is refused, and `RestoreTableFromBackup` reports the backup missing. PostgreSQL and MongoDB complete their copy before returning and expose only `AVAILABLE` | +| Consistency of a backup | Point-in-time snapshot of the table | PostgreSQL holds the table's catalog row shared from the first definition read until the data snapshot (`REPEATABLE READ`, pinned by a read of the table before the row is released) is taken, so the definition recorded is the one in force for the items read, and the items are from one snapshot; the catalog and the data are separate databases, so this is a barrier, not one cross-database snapshot. SQLite reads all items in one deferred read transaction opened while the engine's write lock was held for the definition read, so definition and items agree and writers continue during the copy; that long reader pins WAL checkpointing, so the WAL grows for the duration of the backup by the size of the backup rows written plus any concurrent writes. MongoDB copies with `$out` from the live collection and is not a snapshot | + ## Capacity and Throttling | Area | DynamoDB | ExtendDB | diff --git a/docs/dynamodb-limits.md b/docs/dynamodb-limits.md index b3f4d398f..58f9cbc4b 100755 --- a/docs/dynamodb-limits.md +++ b/docs/dynamodb-limits.md @@ -128,7 +128,10 @@ Source: [AWS DynamoDB Service Quotas](https://docs.aws.amazon.com/amazondynamodb | Limit | DynamoDB Value | Status | Notes | |-------|---------------|--------|-------| -| Concurrent restores | 50 | N/A | ExtendDB does not support backup/restore | +| Concurrent restores | 50 | Partial | 4 per server process on PostgreSQL (a fifth waits up to 30 s, then `LimitExceededException`); SQLite restores and backups copy in batches, each under a short write-lock hold; MongoDB has no limit | A fifth concurrent PostgreSQL restore waits up to 30 s, then fails with `LimitExceededException` | +| Backup contents | Key schema, attribute definitions, items, indexes, billing mode, throughput, table class, SSE | Enforced | Streams, TTL, tags, and deletion protection are not restored, as on the service. Backups are stored inside the catalog database; take database-level backups for off-host copies | +| Point-in-time recovery | 35-day window | N/A | `UpdateContinuousBackups` and `RestoreTableToPointInTime` are refused | +| `RestoreTableFromBackup` overrides | `BillingModeOverride`, `ProvisionedThroughputOverride`, `GlobalSecondaryIndexOverride`, `LocalSecondaryIndexOverride`, `SSESpecificationOverride`, `OnDemandThroughputOverride`, `VectorIndexOverride` | Partial | A request carrying `GlobalSecondaryIndexOverride`, `LocalSecondaryIndexOverride`, `SSESpecificationOverride`, `OnDemandThroughputOverride`, or `VectorIndexOverride` returns `ValidationException` before the target table is created | ## Global Tables @@ -158,7 +161,7 @@ Source: [AWS DynamoDB Service Quotas](https://docs.aws.amazon.com/amazondynamodb | Transactions | 3 | 0 | 1 | 0 | | Streams | 1 | 0 | 3 | 0 | | API-Level | 1 | 0 | 3 | 1 | -| Import/Export/Backup | 0 | 0 | 0 | 8 | +| Import/Export/Backup | 1 | 2 | 0 | 7 | | Global Tables | 0 | 0 | 0 | 2 | | Contributor Insights | 0 | 0 | 0 | 1 | | **Total** | **28** | **1** | **17** | **14** | diff --git a/docs/getting-started.md b/docs/getting-started.md index ffbbc9b3f..7318252e9 100755 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -199,7 +199,7 @@ You should see all checks pass: --- Checking catalog connection... OK: Connected to catalog. --- Checking catalog version... - OK: Catalog version 0.0.3 + OK: Catalog version 0.0.4 --- Checking data database... OK: Connected to data database 'extenddb_catalog_data'. --- Enumerating tables... @@ -215,7 +215,7 @@ extenddb runs as a daemon (background process) and logs to syslog. On startup it ```bash ./target/release/extenddb serve --config extenddb.toml -# extenddb 0.1.13 (catalog 0.0.3) starting on 127.0.0.1:18443 +# extenddb 0.1.13 (catalog 0.0.4) starting on 127.0.0.1:18443 # storage: postgres (postgresql://extenddb:***@localhost:5432/extenddb_catalog) ``` @@ -1294,7 +1294,7 @@ Each runner requires its tools to be installed. The runner checks prerequisites ```bash ./target/release/extenddb version # extenddb 0.1.13 -# catalog 0.0.3 (postgres) +# catalog 0.0.4 (postgres) # commit abc1234 # built 2026-04-17T12:00:00Z ``` diff --git a/docs/manuals/01-architecture-guide.md b/docs/manuals/01-architecture-guide.md index ddcfc6fb8..f6a718b3a 100755 --- a/docs/manuals/01-architecture-guide.md +++ b/docs/manuals/01-architecture-guide.md @@ -172,7 +172,7 @@ extenddb uses a dual-database architecture: - **Catalog database** (e.g., `extenddb_catalog`): Stores table metadata, account/user/group/role/policy definitions, access keys, settings, stream metadata, and metrics. Shared across all accounts. - **Data database** (e.g., `extenddb_catalog_data`): Stores user items, GSI/LSI data, and stream records. Each table gets its own PostgreSQL table. -The catalog version is 0.0.3 on PostgreSQL and SQLite, stored in the `settings` table under the key `catalog_version` and checked at startup. +The catalog version is 0.0.4 on PostgreSQL and SQLite, stored in the `settings` table under the key `catalog_version` and checked at startup. The MongoDB backend tracks its own catalog version, 0.0.2. A mismatch between the compiled-in version and the stored one prevents the server from starting; run `extenddb migrate` to upgrade. diff --git a/docs/manuals/04-quickstart-setup-guide.md b/docs/manuals/04-quickstart-setup-guide.md index b3d8c53f1..cb5926a09 100755 --- a/docs/manuals/04-quickstart-setup-guide.md +++ b/docs/manuals/04-quickstart-setup-guide.md @@ -130,7 +130,7 @@ Check the version: ```bash ./target/release/extenddb version # extenddb 0.1.13 -# catalog 0.0.3 (postgres) +# catalog 0.0.4 (postgres) # commit abc1234 # built 2026-04-17T12:00:00Z ``` @@ -171,7 +171,7 @@ Expected output: --- Checking catalog connection... OK: Connected to catalog. --- Checking catalog version... - OK: Catalog version 0.0.3 + OK: Catalog version 0.0.4 --- Checking data database... OK: Connected to data database 'extenddb_catalog_data'. --- Enumerating tables... diff --git a/docs/manuals/05-admin-guide.md b/docs/manuals/05-admin-guide.md index 90fe53f89..af034d1df 100755 --- a/docs/manuals/05-admin-guide.md +++ b/docs/manuals/05-admin-guide.md @@ -546,10 +546,10 @@ Check that PostgreSQL is running and the connection string in `extenddb.toml` is ``` -Error: catalog version mismatch: found 1.0.0, expected 0.0.3 +Error: catalog version mismatch: found 1.0.0, expected 0.0.4 ``` -Run `extenddb migrate --config extenddb.toml` to upgrade the catalog schema. The check is exact equality in both directions, so this also appears when a binary meets a catalog a newer build already migrated; in that case upgrade the binary rather than the catalog. See the Upgrade Manual for the version history and the stop / migrate / start sequence. +Run `extenddb migrate --config extenddb.toml` to upgrade the catalog schema. The check is exact equality in both directions, so this also appears when a binary meets a catalog a newer build already migrated; in that case upgrade the binary rather than the catalog. `extenddb migrate` refuses a catalog newer than the binary (`catalog version X is newer than this binary's Y`) instead of stamping the older version onto it. See the Upgrade Manual for the version history and the stop / migrate / start sequence. ### Authentication Errors diff --git a/docs/manuals/07-upgrade-manual.md b/docs/manuals/07-upgrade-manual.md index fe2bf5eb2..0390a8051 100755 --- a/docs/manuals/07-upgrade-manual.md +++ b/docs/manuals/07-upgrade-manual.md @@ -4,9 +4,9 @@ ## Current Status -Catalog 0.0.3 is current. The 0.0.2 to 0.0.3 upgrade is the first in-place catalog upgrade ExtendDB has, and **every existing PostgreSQL deployment must run it**, including deployments that never use vector indexes: the server refuses to start against a catalog version it was not built for. +Catalog 0.0.4 is current. **Every existing PostgreSQL and SQLite deployment must run `extenddb migrate`** to reach it: the server refuses to start against a catalog version it was not built for. -See [Catalog 0.0.3](#catalog-003-current) below for what changes and the exact sequence. +See [Catalog 0.0.4](#catalog-004-current) below for what changes and the exact sequence. ## How Catalog Upgrades Work @@ -16,7 +16,8 @@ Migrations are SQL files in `crates/storage-postgres/migrations/`, applied in fi ``` 001_schema.sql ← the complete initial schema -002_vector_indexes.sql ← vector index metadata, catalog 0.0.3 +002_vector_indexes.sql ← vector index metadata, catalog 0.0.3 +003_backup_definitions.sql ← backup table definitions, catalog 0.0.4 ``` The `schema_history` table tracks which files have been applied. When `extenddb migrate` runs, it: @@ -181,7 +182,30 @@ psql -d extenddb_catalog -f catalog_backup_YYYYMMDD.sql ## Version History -### Catalog 0.0.3 (Current) +### Catalog 0.0.4 (Current) + +Adds the table definition a backup records: + +- New `backup_definitions` table: one row per backup, holding the source table's global and local secondary indexes, billing mode and provisioned throughput, table class, and encryption settings. RestoreTableFromBackup recreates them. A backup taken before this upgrade has no row and restores as before, with its keys and items but no secondary indexes. +- New `table_restores` table: one row per table created by RestoreTableFromBackup, naming the backup and the restore time. DescribeTable reports it as `RestoreSummary`, and DeleteBackup is refused with `BackupInUseException` while a restore from the backup is still running. Tables restored before this upgrade have no row and report no `RestoreSummary`. + +Any PostgreSQL table that the old restore bug left in `CREATING` matches the abandoned-restore sweep predicate and is removed on the first control-plane pass after the upgrade. Operators will see those names disappear from `ListTables`; that cleanup is the intended outcome. + +Upgrade sequence, on PostgreSQL and SQLite alike: + +```bash +extenddb stop --config extenddb.toml +extenddb migrate --yes --config extenddb.toml +extenddb serve --config extenddb.toml +``` + +Run `extenddb migrate` without `--yes` first to see what is pending; it reports `catalog 0.0.3 -> 0.0.4` and changes nothing. + +On SQLite, both `migrate` and `serve` acquire the same exclusive lock for the database file, so migration refuses to run while a server is up; stop the server first, as the sequence above does. On startup the SQLite server also removes any restore target left in `CREATING` by a crash, which is only safe because one server at a time can hold the file. A backup keeps a WAL read snapshot open while it copies; checkpoints cannot reclaim pages needed by that snapshot, so allow disk headroom for WAL growth proportional to the backup duration and concurrent write volume. + +The upgrade is not reversible in place: a 0.0.3 binary refuses to start against a 0.0.4 catalog. Roll back by restoring the catalog backup taken before the upgrade, as described above. Backups live in the catalog, so that restore also removes every backup created after the upgrade. + +### Catalog 0.0.3 Adds vector index metadata: diff --git a/docs/manuals/08-install-linux.md b/docs/manuals/08-install-linux.md index 1512201bd..7414ab006 100755 --- a/docs/manuals/08-install-linux.md +++ b/docs/manuals/08-install-linux.md @@ -118,7 +118,7 @@ Expected: ``` === extenddb verify === ... - OK: Catalog version 0.0.3 + OK: Catalog version 0.0.4 ... === HEALTHY: All checks passed === ``` diff --git a/docs/manuals/09-install-macos.md b/docs/manuals/09-install-macos.md index 4e2d87e8d..feec0a409 100755 --- a/docs/manuals/09-install-macos.md +++ b/docs/manuals/09-install-macos.md @@ -96,7 +96,7 @@ Expected: ``` === extenddb verify === ... - OK: Catalog version 0.0.3 + OK: Catalog version 0.0.4 ... === HEALTHY: All checks passed === ``` diff --git a/tests/test_backup_restore_fidelity.py b/tests/test_backup_restore_fidelity.py new file mode 100644 index 000000000..92e2c8595 --- /dev/null +++ b/tests/test_backup_restore_fidelity.py @@ -0,0 +1,656 @@ +# Copyright 2026 ExtendDB contributors +# SPDX-License-Identifier: Apache-2.0 + +"""Backup and restore reproduce the table they were taken from. + +Every backend; hash-only keys and composite keys with S, N, and B sort keys, +plus secondary indexes and provisioned throughput. A restore must reach ACTIVE +with the source's key schema, attribute definitions, and exactly the source's +items, attribute for attribute. (Multi-part base keys are a preview that +restore refuses; that refusal is tested at the storage level.) Comparing whole items rather than counts is what catches a +backend that restores the right number of rows under the wrong keys. + +Readiness assessment P0-2: on PostgreSQL every table with a sort key backed up +without its sort keys, and restoring it left the target in CREATING forever. +""" + +from __future__ import annotations + +import time +import uuid + +import pytest +from botocore.exceptions import ClientError + +from conftest import _poll_interval, wait_for_active, wait_for_deleted + +ITEMS_PER_TABLE = 60 +# The service takes minutes to make a backup AVAILABLE and to restore a table. +BACKUP_TIMEOUT_S = 900.0 +RESTORE_TIMEOUT_S = 1800.0 + + +def _create_backup(client, table_name: str) -> str: + """CreateBackup, then wait for the backup to be AVAILABLE. + + The service creates a backup asynchronously and refuses to restore one + that is still CREATING; ExtendDB returns it AVAILABLE. A just-created + table can also refuse CreateBackup for a short while, which is retried. + """ + deadline = time.monotonic() + 60 + while True: + try: + arn = client.create_backup( + TableName=table_name, BackupName=f"{table_name}-bkp" + )["BackupDetails"]["BackupArn"] + break + except ClientError as e: + code = e.response["Error"]["Code"] + retryable = code in ("ContinuousBackupsUnavailableException", "TableInUseException") + if not retryable or time.monotonic() >= deadline: + raise + time.sleep(0.5) + deadline = time.monotonic() + BACKUP_TIMEOUT_S + try: + while True: + status = client.describe_backup(BackupArn=arn)["BackupDescription"][ + "BackupDetails" + ]["BackupStatus"] + if status == "AVAILABLE": + return arn + assert status == "CREATING", f"backup {arn} is {status}" + assert time.monotonic() < deadline, f"backup {arn} not AVAILABLE in time" + time.sleep(_poll_interval()) + except BaseException: + # The caller never learns the ARN, so the backup is removed here. + try: + client.delete_backup(BackupArn=arn) + except ClientError: + pass + raise + + +def _scan_all(client, table_name: str) -> list[dict]: + items: list[dict] = [] + kwargs: dict = {"TableName": table_name, "ConsistentRead": True} + while True: + resp = client.scan(**kwargs) + items += resp["Items"] + if "LastEvaluatedKey" not in resp: + return items + kwargs["ExclusiveStartKey"] = resp["LastEvaluatedKey"] + + +def _canonical(item: dict) -> str: + """Order-independent, type-preserving form of a wire item for comparison.""" + import json + + def norm(v): + if isinstance(v, dict): + return {k: norm(v[k]) for k in sorted(v)} + if isinstance(v, list): + return [norm(x) for x in v] + if isinstance(v, (bytes, bytearray)): + return {"__bytes__": bytes(v).hex()} + return v + + # Set members have no order on the wire. + def sort_sets(av): + if isinstance(av, dict): + out = {} + for t, val in av.items(): + if t in ("SS", "NS"): + out[t] = sorted(val) + elif t == "BS": + out[t] = sorted(bytes(b).hex() for b in val) + elif t == "M": + out[t] = {k: sort_sets(x) for k, x in val.items()} + elif t == "L": + out[t] = [sort_sets(x) for x in val] + else: + out[t] = val + return out + return av + + return json.dumps(norm({k: sort_sets(v) for k, v in item.items()}), sort_keys=True) + + +def _drop(client, *names: str) -> None: + """Delete tables, waiting out ones still being created or restored.""" + for name in names: + deadline = time.monotonic() + RESTORE_TIMEOUT_S + while True: + try: + client.delete_table(TableName=name) + break + except client.exceptions.ResourceNotFoundException: + break + except client.exceptions.ResourceInUseException: + # The service refuses to delete a table that is CREATING, + # which a restore target is until the restore finishes. + if time.monotonic() >= deadline: + raise + time.sleep(_poll_interval() * 25) + wait_for_deleted(client, name) + + +def _sort_value(kind: str, i: int) -> dict: + if kind == "S": + return {"S": f"sort-{i:04d}"} + if kind == "N": + # Negative, fractional, and large magnitudes in one column. + return {"N": str((i - 30) * 1.5 if i % 3 else (i - 30) * 10**20)} + return {"B": i.to_bytes(2, "big") + b"\x00\xff"} + + +def _hash_value(kind: str, n: int) -> dict: + if kind == "S": + return {"S": f"part-{n}"} + if kind == "N": + return {"N": str(n * 7 - 100)} + return {"B": b"\x00p" + n.to_bytes(2, "big")} + + +def _item(i: int, sort_kind: str | None, hash_kind: str = "S") -> dict: + # With a sort key, several items share a partition; without one, every + # partition key must be distinct or the puts overwrite each other. + item: dict = { + "pk": _hash_value(hash_kind, i % 7 if sort_kind else i), + "str": {"S": f"value-{i}"}, + "num": {"N": str(i)}, + "nested": {"M": {"list": {"L": [{"N": "1"}, {"S": "two"}, {"BOOL": i % 2 == 0}]}}}, + "tags": {"SS": [f"t{i}", "shared"]}, + "blob": {"B": bytes([i % 256]) * 3}, + } + if i % 5 == 0: + item["maybe"] = {"NULL": True} + if sort_kind: + item["sk"] = _sort_value(sort_kind, i) + return item + + +def _query_all(client, **kwargs) -> list[dict]: + items: list[dict] = [] + while True: + resp = client.query(**kwargs) + items += resp["Items"] + if "LastEvaluatedKey" not in resp: + return items + kwargs["ExclusiveStartKey"] = resp["LastEvaluatedKey"] + + +def _index_items(client, table: str, index: str, hash_attr: str, values: list) -> list[str]: + """Canonical items an index serves, across every listed partition. + + Values are strings for an S hash key, or typed attribute values. + """ + out: list[str] = [] + for v in values: + out += [ + _canonical(i) + for i in _query_all( + client, + TableName=table, + IndexName=index, + KeyConditionExpression="#h = :v", + ExpressionAttributeNames={"#h": hash_attr}, + ExpressionAttributeValues={":v": v if isinstance(v, dict) else {"S": v}}, + ) + ] + return sorted(out) + + +def _cleanup(client, backup_arn: str | None, *tables: str) -> None: + """Delete tables and the backup; each step runs even if an earlier one fails.""" + if tables: + try: + _drop(client, tables[0]) + finally: + _cleanup(client, backup_arn, *tables[1:]) + return + if backup_arn: + try: + client.delete_backup(BackupArn=backup_arn) + except client.exceptions.BackupNotFoundException: + pass + + +def _round_trip(client, sort_kind: str | None, hash_kind: str = "S") -> None: + source = f"restore-fid-{uuid.uuid4().hex[:10]}" + restored = f"{source}-r" + key_schema = [{"AttributeName": "pk", "KeyType": "HASH"}] + attr_defs = [{"AttributeName": "pk", "AttributeType": hash_kind}] + if sort_kind: + key_schema.append({"AttributeName": "sk", "KeyType": "RANGE"}) + attr_defs.append({"AttributeName": "sk", "AttributeType": sort_kind}) + backup_arn = None + try: + client.create_table( + TableName=source, + KeySchema=key_schema, + AttributeDefinitions=attr_defs, + BillingMode="PAY_PER_REQUEST", + ) + wait_for_active(client, source) + for i in range(ITEMS_PER_TABLE): + client.put_item(TableName=source, Item=_item(i, sort_kind, hash_kind)) + # Compare against what the source serves, not what was sent: numbers + # come back canonicalized (`-42.0` reads as `-42`), here as in DynamoDB. + expected = _scan_all(client, source) + assert len(expected) == ITEMS_PER_TABLE + + backup_arn = _create_backup(client, source) + resp = client.restore_table_from_backup(TargetTableName=restored, BackupArn=backup_arn) + assert resp["TableDescription"]["TableName"] == restored + started = resp["TableDescription"]["RestoreSummary"] + assert started["SourceBackupArn"] == backup_arn + assert started["RestoreInProgress"] is True + wait_for_active(client, restored, timeout=RESTORE_TIMEOUT_S) + + table = client.describe_table(TableName=restored)["Table"] + # DescribeTable keeps reporting where the table came from, finished. + summary = table["RestoreSummary"] + assert summary["SourceBackupArn"] == backup_arn + assert summary["RestoreInProgress"] is False + assert summary["RestoreDateTime"] == started["RestoreDateTime"] + assert table["KeySchema"] == key_schema + assert sorted(table["AttributeDefinitions"], key=lambda a: a["AttributeName"]) == sorted( + attr_defs, key=lambda a: a["AttributeName"] + ) + assert table["BillingModeSummary"]["BillingMode"] == "PAY_PER_REQUEST" + + got = sorted(_canonical(i) for i in _scan_all(client, restored)) + want = sorted(_canonical(i) for i in expected) + assert len(got) == len(want), f"restored {len(got)} of {len(want)} items" + assert got == want + + # Point reads by full key, so a restore that stored the right items + # under the wrong key columns fails here even if Scan looked right. + for item in expected[:10]: + key = {"pk": item["pk"]} + if sort_kind: + key["sk"] = item["sk"] + fetched = client.get_item(TableName=restored, Key=key, ConsistentRead=True) + assert _canonical(fetched.get("Item", {})) == _canonical(item) + + # The restored table takes writes like any other. + client.put_item( + TableName=restored, Item=_item(ITEMS_PER_TABLE, sort_kind, hash_kind) + ) + finally: + _cleanup(client, backup_arn, restored, source) + + +@pytest.mark.parametrize("hash_kind", ["S", "N", "B"]) +def test_restore_hash_only_table(dynamodb_client, hash_kind): + _round_trip(dynamodb_client, None, hash_kind) + + +@pytest.mark.parametrize("sort_kind", ["S", "N", "B"]) +def test_restore_composite_key_table(dynamodb_client, sort_kind): + _round_trip(dynamodb_client, sort_kind) + + +def _gsi(name: str, hash_attr: str, range_attr: str | None, projection: dict, **extra) -> dict: + ks = [{"AttributeName": hash_attr, "KeyType": "HASH"}] + if range_attr: + ks.append({"AttributeName": range_attr, "KeyType": "RANGE"}) + return {"IndexName": name, "KeySchema": ks, "Projection": projection, **extra} + + +def _index_by_name(indexes: list[dict]) -> dict: + return {i["IndexName"]: i for i in indexes} + + +def test_restore_preserves_indexes_and_provisioned_throughput(dynamodb_client): + """GSIs and LSIs, their key schemas, projections, and throughput, the + table's provisioned throughput, and every index's contents come back.""" + client = dynamodb_client + source = f"restore-idx-{uuid.uuid4().hex[:10]}" + restored = f"{source}-r" + gsi_tp = {"ReadCapacityUnits": 3, "WriteCapacityUnits": 4} + gsis = [ + _gsi("by_owner", "owner", "rank", {"ProjectionType": "ALL"}, ProvisionedThroughput=gsi_tp), + _gsi( + "by_kind", + "kind", + None, + {"ProjectionType": "INCLUDE", "NonKeyAttributes": ["note"]}, + ProvisionedThroughput=gsi_tp, + ), + _gsi("by_kind_keys", "kind", "rank", {"ProjectionType": "KEYS_ONLY"}, + ProvisionedThroughput=gsi_tp), + _gsi("by_code", "code", "tag", {"ProjectionType": "ALL"}, + ProvisionedThroughput=gsi_tp), + ] + lsis = [ + { + "IndexName": "by_rank", + "KeySchema": [ + {"AttributeName": "pk", "KeyType": "HASH"}, + {"AttributeName": "rank", "KeyType": "RANGE"}, + ], + "Projection": {"ProjectionType": "ALL"}, + } + ] + backup_arn = None + try: + client.create_table( + TableName=source, + KeySchema=[ + {"AttributeName": "pk", "KeyType": "HASH"}, + {"AttributeName": "sk", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "pk", "AttributeType": "S"}, + {"AttributeName": "sk", "AttributeType": "S"}, + {"AttributeName": "owner", "AttributeType": "S"}, + {"AttributeName": "kind", "AttributeType": "S"}, + {"AttributeName": "rank", "AttributeType": "N"}, + {"AttributeName": "code", "AttributeType": "N"}, + {"AttributeName": "tag", "AttributeType": "B"}, + ], + ProvisionedThroughput={"ReadCapacityUnits": 7, "WriteCapacityUnits": 9}, + GlobalSecondaryIndexes=gsis, + LocalSecondaryIndexes=lsis, + ) + wait_for_active(client, source) + owners = [f"owner-{i}" for i in range(3)] + kinds = ["red", "blue"] + for i in range(ITEMS_PER_TABLE): + item = { + "pk": {"S": f"part-{i % 4}"}, + "sk": {"S": f"sort-{i:04d}"}, + # Unique, so every ordered index read is fully determined. + "rank": {"N": str(i * 3 - 50)}, + "code": {"N": str(i % 5 - 2)}, + "tag": {"B": bytes([255 - i, i])}, + "note": {"S": f"note-{i}"}, + "other": {"S": "not projected into by_kind"}, + } + # Sparse: some items lack one index key or the other. + if i % 3: + item["owner"] = {"S": owners[i % 3]} + if i % 4: + item["kind"] = {"S": kinds[i % 2]} + client.put_item(TableName=source, Item=item) + + # What each index serves on the source, read after the source's GSIs + # have caught up with the writes. + def index_view(table: str) -> dict: + return { + "by_owner": _index_items(client, table, "by_owner", "owner", owners), + "by_kind": _index_items(client, table, "by_kind", "kind", kinds), + "by_kind_keys": _index_items(client, table, "by_kind_keys", "kind", kinds), + "by_rank": _index_items(client, table, "by_rank", "pk", + [f"part-{p}" for p in range(4)]), + "by_code": _index_items(client, table, "by_code", "code", + [{"N": str(c)} for c in range(-2, 3)]), + } + + expected_sizes = { + "by_owner": sum(1 for i in range(ITEMS_PER_TABLE) if i % 3), + "by_kind": sum(1 for i in range(ITEMS_PER_TABLE) if i % 4), + "by_kind_keys": sum(1 for i in range(ITEMS_PER_TABLE) if i % 4), + "by_rank": ITEMS_PER_TABLE, + "by_code": ITEMS_PER_TABLE, + } + deadline = time.monotonic() + 120 + while True: + want = index_view(source) + sizes = {k: len(v) for k, v in want.items()} + if sizes == expected_sizes: + break + assert time.monotonic() < deadline, f"source GSIs did not converge: {sizes}" + time.sleep(_poll_interval() * 10) + + backup_arn = _create_backup(client, source) + client.restore_table_from_backup(TargetTableName=restored, BackupArn=backup_arn) + wait_for_active(client, restored, timeout=RESTORE_TIMEOUT_S) + + table = client.describe_table(TableName=restored)["Table"] + assert table["ProvisionedThroughput"]["ReadCapacityUnits"] == 7 + assert table["ProvisionedThroughput"]["WriteCapacityUnits"] == 9 + got_gsis = _index_by_name(table.get("GlobalSecondaryIndexes", [])) + assert sorted(got_gsis) == sorted(g["IndexName"] for g in gsis) + for g in gsis: + r = got_gsis[g["IndexName"]] + assert r["KeySchema"] == g["KeySchema"] + assert r["Projection"] == g["Projection"] + assert r["ProvisionedThroughput"]["ReadCapacityUnits"] == 3 + assert r["ProvisionedThroughput"]["WriteCapacityUnits"] == 4 + for g in gsis: + assert got_gsis[g["IndexName"]]["IndexStatus"] == "ACTIVE" + got_lsis = _index_by_name(table.get("LocalSecondaryIndexes", [])) + assert sorted(got_lsis) == ["by_rank"] + assert got_lsis["by_rank"]["KeySchema"] == lsis[0]["KeySchema"] + assert got_lsis["by_rank"]["Projection"] == lsis[0]["Projection"] + + # The service backfills GSIs after the table turns ACTIVE; poll. + deadline = time.monotonic() + RESTORE_TIMEOUT_S + while True: + got = index_view(restored) + if got == want: + break + assert time.monotonic() < deadline, { + k: (len(got[k]), len(want[k])) for k in want + } + time.sleep(_poll_interval() * 10) + + # Projections checked against their definitions, not just against the + # source table, so a projection bug both tables share still fails. + def attr_names(index: str, hash_attr: str, value: dict) -> set[frozenset]: + return { + frozenset(i) + for i in _query_all( + client, + TableName=restored, + IndexName=index, + KeyConditionExpression="#h = :v", + ExpressionAttributeNames={"#h": hash_attr}, + ExpressionAttributeValues={":v": value}, + ) + } + + assert attr_names("by_kind", "kind", {"S": "blue"}) == { + frozenset({"pk", "sk", "kind", "note"}) + } + assert attr_names("by_kind_keys", "kind", {"S": "blue"}) == { + frozenset({"pk", "sk", "kind", "rank"}) + } + all_attrs = attr_names("by_owner", "owner", {"S": owners[1]}) + assert all_attrs and all({"other", "note", "rank", "tag"} <= a for a in all_attrs) + + # Ordered, ranged reads on a numeric index sort key come back in the + # same order from both tables, in both directions. + for forward in (True, False): + def ranked(table: str) -> list[str]: + return [ + _canonical(i) + for i in _query_all( + client, + TableName=table, + IndexName="by_owner", + KeyConditionExpression="#o = :o AND #r BETWEEN :lo AND :hi", + ExpressionAttributeNames={"#o": "owner", "#r": "rank"}, + ExpressionAttributeValues={ + ":o": {"S": owners[1]}, + ":lo": {"N": "-20"}, + ":hi": {"N": "100"}, + }, + ScanIndexForward=forward, + ) + ] + + src_order = ranked(source) + assert src_order + assert ranked(restored) == src_order + finally: + _cleanup(client, backup_arn, restored, source) + + +def test_restore_normalizes_stale_throughput_after_on_demand_switch(dynamodb_client): + """A backup taken after an on-demand switch does not restore stale capacity.""" + client = dynamodb_client + source = f"restore-ondemand-{uuid.uuid4().hex[:10]}" + restored = f"{source}-r" + backup_arn = None + try: + client.create_table( + TableName=source, + KeySchema=[{"AttributeName": "pk", "KeyType": "HASH"}], + AttributeDefinitions=[ + {"AttributeName": "pk", "AttributeType": "S"}, + {"AttributeName": "gpk", "AttributeType": "S"}, + ], + BillingMode="PROVISIONED", + ProvisionedThroughput={"ReadCapacityUnits": 7, "WriteCapacityUnits": 9}, + GlobalSecondaryIndexes=[ + _gsi( + "by_gpk", + "gpk", + None, + {"ProjectionType": "ALL"}, + ProvisionedThroughput={ + "ReadCapacityUnits": 3, + "WriteCapacityUnits": 4, + }, + ) + ], + ) + wait_for_active(client, source) + client.update_table(TableName=source, BillingMode="PAY_PER_REQUEST") + wait_for_active(client, source) + + backup_arn = _create_backup(client, source) + client.restore_table_from_backup(TargetTableName=restored, BackupArn=backup_arn) + wait_for_active(client, restored, timeout=RESTORE_TIMEOUT_S) + + table = client.describe_table(TableName=restored)["Table"] + assert table["BillingModeSummary"]["BillingMode"] == "PAY_PER_REQUEST" + assert table["ProvisionedThroughput"]["ReadCapacityUnits"] == 0 + assert table["ProvisionedThroughput"]["WriteCapacityUnits"] == 0 + gsi = _index_by_name(table["GlobalSecondaryIndexes"])["by_gpk"] + assert gsi["ProvisionedThroughput"]["ReadCapacityUnits"] == 0 + assert gsi["ProvisionedThroughput"]["WriteCapacityUnits"] == 0 + finally: + _cleanup(client, backup_arn, restored, source) + + +def test_restore_override_pay_per_request_to_provisioned(dynamodb_client): + client = dynamodb_client + source = f"restore-override-ppr-{uuid.uuid4().hex[:10]}" + restored = f"{source}-r" + backup_arn = None + try: + client.create_table( + TableName=source, + KeySchema=[{"AttributeName": "pk", "KeyType": "HASH"}], + AttributeDefinitions=[{"AttributeName": "pk", "AttributeType": "S"}], + BillingMode="PAY_PER_REQUEST", + ) + wait_for_active(client, source) + + backup_arn = _create_backup(client, source) + client.restore_table_from_backup( + TargetTableName=restored, + BackupArn=backup_arn, + BillingModeOverride="PROVISIONED", + ProvisionedThroughputOverride={ + "ReadCapacityUnits": 5, + "WriteCapacityUnits": 5, + }, + ) + wait_for_active(client, restored, timeout=RESTORE_TIMEOUT_S) + + table = client.describe_table(TableName=restored)["Table"] + # BillingModeSummary is omitted for a provisioned table by both the + # service and ExtendDB; omission therefore means PROVISIONED here. + assert table.get("BillingModeSummary", {}).get("BillingMode", "PROVISIONED") == ( + "PROVISIONED" + ) + assert table["ProvisionedThroughput"]["ReadCapacityUnits"] == 5 + assert table["ProvisionedThroughput"]["WriteCapacityUnits"] == 5 + finally: + _cleanup(client, backup_arn, restored, source) + + +def test_restore_override_provisioned_to_pay_per_request(dynamodb_client): + client = dynamodb_client + source = f"restore-override-prov-{uuid.uuid4().hex[:10]}" + restored = f"{source}-r" + backup_arn = None + try: + client.create_table( + TableName=source, + KeySchema=[{"AttributeName": "pk", "KeyType": "HASH"}], + AttributeDefinitions=[{"AttributeName": "pk", "AttributeType": "S"}], + BillingMode="PROVISIONED", + ProvisionedThroughput={ + "ReadCapacityUnits": 7, + "WriteCapacityUnits": 9, + }, + ) + wait_for_active(client, source) + + backup_arn = _create_backup(client, source) + client.restore_table_from_backup( + TargetTableName=restored, + BackupArn=backup_arn, + BillingModeOverride="PAY_PER_REQUEST", + ) + wait_for_active(client, restored, timeout=RESTORE_TIMEOUT_S) + + table = client.describe_table(TableName=restored)["Table"] + assert table["BillingModeSummary"]["BillingMode"] == "PAY_PER_REQUEST" + assert table["ProvisionedThroughput"]["ReadCapacityUnits"] == 0 + assert table["ProvisionedThroughput"]["WriteCapacityUnits"] == 0 + finally: + _cleanup(client, backup_arn, restored, source) + + +def test_provisioned_restore_override_requiring_gsi_override_leaves_no_target( + dynamodb_client, +): + client = dynamodb_client + source = f"restore-override-gsi-{uuid.uuid4().hex[:10]}" + restored = f"{source}-r" + backup_arn = None + try: + client.create_table( + TableName=source, + KeySchema=[{"AttributeName": "pk", "KeyType": "HASH"}], + AttributeDefinitions=[ + {"AttributeName": "pk", "AttributeType": "S"}, + {"AttributeName": "gpk", "AttributeType": "S"}, + ], + BillingMode="PAY_PER_REQUEST", + GlobalSecondaryIndexes=[ + _gsi("by_gpk", "gpk", None, {"ProjectionType": "ALL"}) + ], + ) + wait_for_active(client, source) + backup_arn = _create_backup(client, source) + + with pytest.raises(ClientError) as exc_info: + client.restore_table_from_backup( + TargetTableName=restored, + BackupArn=backup_arn, + BillingModeOverride="PROVISIONED", + ProvisionedThroughputOverride={ + "ReadCapacityUnits": 5, + "WriteCapacityUnits": 5, + }, + ) + error = exc_info.value.response["Error"] + assert error["Code"] == "ValidationException" + assert error["Message"] == ( + "One or more parameter values were invalid: " + "GlobalSecondaryIndexOverride must be specified for index: by_gpk " + "when BillingModeOverride is PROVISIONED" + ) + + with pytest.raises(client.exceptions.ResourceNotFoundException): + client.describe_table(TableName=restored) + finally: + _cleanup(client, backup_arn, restored, source) diff --git a/tests/test_cli_vector_catalog_migration.py b/tests/test_cli_vector_catalog_migration.py index 1b260e680..e50f9f175 100644 --- a/tests/test_cli_vector_catalog_migration.py +++ b/tests/test_cli_vector_catalog_migration.py @@ -1,7 +1,10 @@ # Copyright 2026 ExtendDB contributors # SPDX-License-Identifier: Apache-2.0 -"""Catalog migration tests for the vector index schema (catalog 0.0.2 -> 0.0.3). +"""Catalog migration tests against a live PostgreSQL deployment. + +The vector index schema (catalog 0.0.3) and the backup table definitions +(catalog 0.0.4) are exercised through an upgrade from the pre-vector shape. These exercise the migration against a live deployment rather than asserting the SQL text: init one, roll its catalog back to the pre-vector shape, and check that @@ -29,6 +32,14 @@ ) VECTOR_MIGRATION = "002_vector_indexes.sql" +BACKUP_DEFINITIONS_MIGRATION = "003_backup_definitions.sql" + +# The version the binary expects, written by the last catalog migration. Every +# current-version assertion reads it from here so the next bump is one edit. +CURRENT_CATALOG_VERSION = "0.0.4" + +# A version no release has reached, for the symmetric version-gate test. +FUTURE_CATALOG_VERSION = "0.0.5" def _catalog_conn(cli_env): @@ -67,6 +78,12 @@ def _backups_has_vector_column(cli_env): ) +def _backup_definitions_table_exists(cli_env): + return _catalog_query( + cli_env, "SELECT to_regclass('public.backup_definitions') IS NOT NULL" + ) + + def _init(cli_env): result = _run_extenddb( "init", *_init_args(cli_env), @@ -80,17 +97,19 @@ def _roll_catalog_back_to_pre_vector(cli_env): """Turn a freshly initialised catalog into the shape 0.0.2 deployments have. Reproduces an upgrade rather than a fresh install, which is the case that - matters: a fresh install applies both migrations in order and reaches the same + matters: a fresh install applies the migrations in order and reaches the same end state trivially. """ conn = _catalog_conn(cli_env) try: conn.autocommit = True with conn.cursor() as cur: + cur.execute("DROP TABLE IF EXISTS backup_definitions") cur.execute("DROP TABLE IF EXISTS vector_indexes") cur.execute("ALTER TABLE backups DROP COLUMN IF EXISTS vector_indexes") cur.execute( - "DELETE FROM schema_history WHERE filename = %s", (VECTOR_MIGRATION,) + "DELETE FROM schema_history WHERE filename IN (%s, %s)", + (VECTOR_MIGRATION, BACKUP_DEFINITIONS_MIGRATION), ) cur.execute("UPDATE settings SET value = '0.0.2' WHERE key = 'catalog_version'") finally: @@ -98,10 +117,10 @@ def _roll_catalog_back_to_pre_vector(cli_env): class TestVectorCatalogMigration: - """Catalog version 0.0.3: the vector index table and the backup snapshot.""" + """Migrations from the pre-vector catalog shape to the current version.""" - def test_init_creates_the_vector_catalog_at_0_0_3(self, cli_env): - """A fresh init applies both catalog migrations and records the version. + def test_init_creates_the_catalog_at_the_current_version(self, cli_env): + """A fresh init applies every catalog migration and records the version. The version is what the binary checks at startup, so a migration that creates the table without moving the version, or the reverse, would leave @@ -109,9 +128,10 @@ def test_init_creates_the_vector_catalog_at_0_0_3(self, cli_env): """ _init(cli_env) - assert _catalog_version(cli_env) == "0.0.3" + assert _catalog_version(cli_env) == CURRENT_CATALOG_VERSION assert _vector_table_exists(cli_env) is True assert _backups_has_vector_column(cli_env) is True + assert _backup_definitions_table_exists(cli_env) is True conn = _catalog_conn(cli_env) try: @@ -124,6 +144,7 @@ def test_init_creates_the_vector_catalog_at_0_0_3(self, cli_env): # deployment that is already current. The statements are idempotent too, # so a replay after a crash between applying and recording is harmless. assert VECTOR_MIGRATION in tracked, tracked + assert BACKUP_DEFINITIONS_MIGRATION in tracked, tracked def test_migrate_upgrades_a_pre_vector_deployment(self, cli_env): """A 0.0.2 deployment is refused, upgraded by migrate, and then serves.""" @@ -148,7 +169,7 @@ def test_migrate_upgrades_a_pre_vector_deployment(self, cli_env): ) from None combined = refused_serve.stdout + refused_serve.stderr assert refused_serve.returncode != 0, combined - assert "0.0.3" in combined and "0.0.2" in combined, combined + assert CURRENT_CATALOG_VERSION in combined and "0.0.2" in combined, combined # Without --yes, migrate reports the pending upgrade and changes nothing. pending = _run_extenddb( @@ -156,7 +177,7 @@ def test_migrate_upgrades_a_pre_vector_deployment(self, cli_env): ) pending_output = pending.stdout + pending.stderr assert pending.returncode != 0, pending_output - assert "0.0.2 -> 0.0.3" in pending_output, pending_output + assert f"0.0.2 -> {CURRENT_CATALOG_VERSION}" in pending_output, pending_output assert _vector_table_exists(cli_env) is False assert _catalog_version(cli_env) == "0.0.2" @@ -164,7 +185,7 @@ def test_migrate_upgrades_a_pre_vector_deployment(self, cli_env): "migrate", "--yes", *_pg_args(), config=cli_env["config_path"], check=False ) assert applied.returncode == 0, applied.stdout + applied.stderr - assert _catalog_version(cli_env) == "0.0.3" + assert _catalog_version(cli_env) == CURRENT_CATALOG_VERSION assert _vector_table_exists(cli_env) is True assert _backups_has_vector_column(cli_env) is True @@ -235,7 +256,7 @@ def test_migrate_survives_an_applied_but_unrecorded_migration(self, cli_env): assert f"Migration {VECTOR_MIGRATION} failed" not in output, output # The replay leaves the same end state, and the ledger is repaired. - assert _catalog_version(cli_env) == "0.0.3" + assert _catalog_version(cli_env) == CURRENT_CATALOG_VERSION assert _vector_table_exists(cli_env) is True assert _backups_has_vector_column(cli_env) is True conn = _catalog_conn(cli_env) @@ -270,7 +291,8 @@ def test_serve_refuses_a_catalog_newer_than_the_binary(self, cli_env): conn.autocommit = True with conn.cursor() as cur: cur.execute( - "UPDATE settings SET value = '0.0.4' WHERE key = 'catalog_version'" + "UPDATE settings SET value = %s WHERE key = 'catalog_version'", + (FUTURE_CATALOG_VERSION,), ) finally: conn.close() @@ -285,8 +307,11 @@ def test_serve_refuses_a_catalog_newer_than_the_binary(self, cli_env): except subprocess.TimeoutExpired: _run_extenddb("stop", config=cli_env["config_path"], check=False) raise AssertionError( - "serve started against a 0.0.4 catalog; the version gate is not symmetric" + f"serve started against a {FUTURE_CATALOG_VERSION} catalog; " + "the version gate is not symmetric" ) from None combined = refused.stdout + refused.stderr assert refused.returncode != 0, combined - assert "0.0.3" in combined and "0.0.4" in combined, combined + assert CURRENT_CATALOG_VERSION in combined and FUTURE_CATALOG_VERSION in combined, ( + combined + )