Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

4 changes: 4 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -136,6 +136,10 @@ extenddb healthcheck --endpoint https://127.0.0.1:18443 # explicit target

This is a liveness check, which is what a container `HEALTHCHECK` wants: `/health` does not query the storage backend, so it reports healthy even if PostgreSQL becomes unreachable after startup. That is deliberate, since a liveness probe that failed on a database outage would restart every replica at once. A backend that is unreachable at startup does stop the server from listening, so that case is caught. There is no separate readiness endpoint yet.

### One instance per database

Run one `extenddb serve` per catalog. On SQLite this is enforced: the server takes an exclusive lock on `<database>.lock` next to the database file and a second server on the same file refuses to start (`init`, `migrate`, and `destroy` take the same lock while they run). On PostgreSQL and MongoDB nothing prevents a second instance, but running more than one against the same catalog is not supported in this release: every instance runs every background worker, credential and policy caches are per instance with no cross-instance invalidation (a revoked key stays valid on the other instance for up to the cache TTL), and `/health` does not check the database. See the [deployment guide](docs/manuals/11-deployment-guide.md#multi-instance-considerations).

To make the generated self-signed certificate valid for the name clients use — an in-cluster service DNS name, for example — pass `--tls-san` to `init` (repeatable):

```bash
Expand Down
1 change: 1 addition & 0 deletions crates/storage-sqlite/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -35,3 +35,4 @@ rand = { workspace = true }
bigdecimal = { workspace = true }
crc32fast = { workspace = true }
zeroize = { workspace = true }
libc = { workspace = true }
113 changes: 112 additions & 1 deletion crates/storage-sqlite/src/bootstrapper.rs
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ use sqlx::SqlitePool;
use sqlx::sqlite::SqlitePoolOptions;

use crate::schema::{self, CATALOG_VERSION};
use crate::serve_lock::ServeLock;
use crate::sqlite_util::sqlite_url;

/// SQLite backend bootstrapper.
Expand All @@ -29,11 +30,41 @@ use crate::sqlite_util::sqlite_url;
/// paths, not the hot serving path.
pub struct SqliteBootstrapper {
path: String,
/// The database-file lock held between [`Bootstrapper::acquire_migration_lock`]
/// and [`Bootstrapper::release_migration_lock`]; see `serve_lock`.
held_lock: std::sync::Mutex<Option<ServeLock>>,
}

impl SqliteBootstrapper {
pub fn new(path: impl Into<String>) -> Self {
Self { path: path.into() }
Self {
path: path.into(),
held_lock: std::sync::Mutex::new(None),
}
}

/// Take the exclusive lock a running `extenddb serve` holds on this
/// database, or explain which process holds it. `Ok(None)` for an
/// in-memory database.
fn exclusive_lock(&self, what: &str) -> OpResult<Option<ServeLock>> {
// The lock file lives next to the database; `init` may be the one
// creating that directory (see `pool`).
if !self.is_memory()
&& let Some(parent) = std::path::Path::new(&self.path).parent()
&& !parent.as_os_str().is_empty()
{
std::fs::create_dir_all(parent).map_err(|e| {
OpError::Internal(format!(
"create parent directory for SQLite database '{}': {e}",
self.path
))
})?;
}
ServeLock::acquire_for_location(&self.path).map_err(|e| {
OpError::Internal(format!(
"{what} needs exclusive use of the SQLite database and cannot have it: {e}"
))
})
}

/// Build a `SqliteBootstrapper` from the config file and CLI args.
Expand Down Expand Up @@ -444,10 +475,37 @@ impl Bootstrapper for SqliteBootstrapper {
Ok(row.map(|(v,)| v))
}

/// `init` and `migrate` rewrite the schema, so they take the same
/// exclusive lock `extenddb serve` holds on the database file. Unlike the
/// trait's default contract this does not wait for the lock: the holder is
/// a server that runs until stopped, so waiting would hang, and the
/// operator is told to stop it instead.
async fn acquire_migration_lock(&self) -> OpResult<()> {
let lock = self.exclusive_lock("this command")?;
*self
.held_lock
.lock()
.map_err(|_| OpError::Internal("lock state poisoned".to_owned()))? = lock;
Ok(())
}

async fn release_migration_lock(&self) -> OpResult<()> {
self.held_lock
.lock()
.map_err(|_| OpError::Internal("lock state poisoned".to_owned()))?
.take();
Ok(())
}

async fn drop_databases(&self, _data_db: &str) -> OpResult<()> {
if self.is_memory() {
return Ok(());
}
// A server that still has the file open would keep serving an unlinked
// database, and a later `init` would create a second one at the same
// path. Refuse while any other process holds the file, and hold it
// ourselves until the files are gone.
let _lock = self.exclusive_lock("destroy")?;
if std::path::Path::new(&self.path).exists() {
std::fs::remove_file(&self.path)
.map_err(|e| OpError::Internal(format!("remove database file: {e}")))?;
Expand Down Expand Up @@ -498,3 +556,56 @@ impl Bootstrapper for SqliteBootstrapper {
)
}
}

#[cfg(all(test, unix))]
mod lock_tests {
use super::SqliteBootstrapper;
use crate::serve_lock::ServeLock;
use extenddb_storage::bootstrapper::Bootstrapper;

/// `destroy` and `migrate` are refused while a server holds the database,
/// and a server is refused while `migrate` holds it.
#[tokio::test]
async fn destroy_and_migrate_are_refused_while_a_server_holds_the_file() {
let dir = std::env::temp_dir().join(format!(
"extenddb-bootstrap-lock-{}",
uuid::Uuid::new_v4().simple()
));
std::fs::create_dir_all(&dir).expect("dir");
let db = dir.join("db.sqlite");
std::fs::write(&db, b"").expect("db file");
let bootstrap = SqliteBootstrapper::new(db.to_string_lossy().into_owned());

let server = ServeLock::acquire(&db).expect("server lock");
let err = bootstrap
.drop_databases("unused")
.await
.expect_err("destroy under a server");
assert!(
format!("{err:?}").contains("another extenddb process"),
"{err:?}"
);
assert!(
db.exists(),
"destroy must not unlink a file a server has open"
);
let err = bootstrap
.acquire_migration_lock()
.await
.expect_err("migrate under a server");
assert!(
format!("{err:?}").contains("another extenddb process"),
"{err:?}"
);
drop(server);

bootstrap.acquire_migration_lock().await.expect("free now");
assert!(ServeLock::acquire(&db).is_err(), "serve during migrate");
bootstrap.release_migration_lock().await.expect("release");
ServeLock::acquire(&db).expect("free after release");

bootstrap.drop_databases("unused").await.expect("destroy");
assert!(!db.exists());
let _ = std::fs::remove_dir_all(&dir);
}
}
50 changes: 49 additions & 1 deletion crates/storage-sqlite/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,7 @@ mod metadata;
mod number_key;
mod operations;
mod schema;
mod serve_lock;
mod sqlite_util;
mod store;
mod stream;
Expand Down Expand Up @@ -188,7 +189,16 @@ fn sqlite_server_components_factory(
let pool_size = config.max_connections();
let region = region.to_owned();
Box::pin(async move {
let engine = SqliteEngine::new(&db_path, pool_size, &region, MAX_ITEM_SIZE_BYTES)
// One server per database file, taken before the database is opened:
// startup recovery below assumes no other server is using it. A file
// database only; an in-memory one belongs to this process alone.
let serve_lock = serve_lock::ServeLock::acquire_for_location(&db_path)
.map_err(BackendError::InitializationFailed)?
.map(|lock| {
tracing::debug!("holding serve lock {}", lock.path().display());
std::sync::Arc::new(lock)
});
let mut engine = SqliteEngine::new(&db_path, pool_size, &region, MAX_ITEM_SIZE_BYTES)
.await
.map_err(|e| BackendError::ConnectionFailed {
backend: "sqlite".to_owned(),
Expand Down Expand Up @@ -259,6 +269,7 @@ fn sqlite_server_components_factory(
Err(e) => tracing::error!("Failed to reconcile incomplete vector indexes: {e}"),
}

engine.serve_lock = serve_lock;
let control_plane_notify = engine.control_plane_notify();
let engine = Arc::new(engine);

Expand Down Expand Up @@ -304,3 +315,40 @@ fn sqlite_server_components_factory(
})
})
}

#[cfg(all(test, unix))]
mod serve_lock_tests {
/// The server factory itself takes the lock: a second server on the same
/// file fails to start while the first is up, and starts once it is gone.
#[tokio::test]
async fn a_second_server_on_one_file_is_refused() {
let dir = std::env::temp_dir().join(format!(
"extenddb-serve-lock-factory-{}",
uuid::Uuid::new_v4().simple()
));
std::fs::create_dir_all(&dir).expect("dir");
let config = crate::config::SqliteConfig {
path: dir.join("db.sqlite").to_string_lossy().into_owned(),
pool_size: 2,
};
let mut options = extenddb_storage::server_components::ServerComponentsOptions::default();
options.bootstrap_if_uninitialized = true;
let first = super::sqlite_server_components_factory(&config, "us-east-1", options)
.await
.expect("first server");
let Err(err) = super::sqlite_server_components_factory(&config, "us-east-1", options).await
else {
panic!("a second server on the same file started");
};
assert!(
err.to_string().contains("another extenddb process"),
"{err}"
);
drop(first);
let again = super::sqlite_server_components_factory(&config, "us-east-1", options)
.await
.expect("starts once the first is gone");
drop(again);
let _ = std::fs::remove_dir_all(&dir);
}
}
Loading
Loading