diff --git a/.gitignore b/.gitignore index 0c048b6..2d1a1d3 100644 --- a/.gitignore +++ b/.gitignore @@ -4,3 +4,6 @@ !.env.example *.log __pycache__/ + +# Local benchmark reports, logs and fixtures. +/benchmarks/ diff --git a/Cargo.lock b/Cargo.lock index 3e53454..746304e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -739,6 +739,7 @@ dependencies = [ "futures-util", "getrandom 0.3.4", "object_store", + "prost", "reqwest 0.12.28", "serde", "serde_json", @@ -860,7 +861,7 @@ dependencies = [ [[package]] name = "cellule-app" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" +source = "git+https://github.com/crabbuild/cellule.git?rev=1d0648b3b5ae0cca040c614b505b4d72bb769ab1#1d0648b3b5ae0cca040c614b505b4d72bb769ab1" dependencies = [ "blake3", "cellule-runtime", @@ -869,7 +870,7 @@ dependencies = [ [[package]] name = "cellule-host" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" +source = "git+https://github.com/crabbuild/cellule.git?rev=1d0648b3b5ae0cca040c614b505b4d72bb769ab1#1d0648b3b5ae0cca040c614b505b4d72bb769ab1" dependencies = [ "cellule-app", "cellule-runtime", @@ -883,7 +884,7 @@ dependencies = [ [[package]] name = "cellule-ltx" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" +source = "git+https://github.com/crabbuild/cellule.git?rev=1d0648b3b5ae0cca040c614b505b4d72bb769ab1#1d0648b3b5ae0cca040c614b505b4d72bb769ab1" dependencies = [ "async-trait", "blake3", @@ -905,7 +906,7 @@ dependencies = [ [[package]] name = "cellule-peer-http" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" +source = "git+https://github.com/crabbuild/cellule.git?rev=1d0648b3b5ae0cca040c614b505b4d72bb769ab1#1d0648b3b5ae0cca040c614b505b4d72bb769ab1" dependencies = [ "axum", "cellule-runtime", @@ -927,7 +928,7 @@ dependencies = [ [[package]] name = "cellule-runtime" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" +source = "git+https://github.com/crabbuild/cellule.git?rev=1d0648b3b5ae0cca040c614b505b4d72bb769ab1#1d0648b3b5ae0cca040c614b505b4d72bb769ab1" dependencies = [ "blake3", "bytes", @@ -953,7 +954,7 @@ dependencies = [ [[package]] name = "cellule-store" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" +source = "git+https://github.com/crabbuild/cellule.git?rev=1d0648b3b5ae0cca040c614b505b4d72bb769ab1#1d0648b3b5ae0cca040c614b505b4d72bb769ab1" dependencies = [ "async-trait", "blake3", @@ -976,7 +977,7 @@ dependencies = [ [[package]] name = "cellule-types" version = "0.1.0" -source = "git+https://github.com/crabbuild/cellule.git?rev=9e17746a633ca1046bd93866866074226091f81a#9e17746a633ca1046bd93866866074226091f81a" +source = "git+https://github.com/crabbuild/cellule.git?rev=1d0648b3b5ae0cca040c614b505b4d72bb769ab1#1d0648b3b5ae0cca040c614b505b4d72bb769ab1" dependencies = [ "schemars", "serde", @@ -3269,6 +3270,7 @@ dependencies = [ "bytes", "futures-core", "futures-util", + "h2 0.4.19", "http 1.5.0", "http-body 1.1.0", "http-body-util", diff --git a/Cargo.toml b/Cargo.toml index fb94094..c7d012b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -8,12 +8,12 @@ repository = "https://github.com/crabbuild/beyonddb" rust-version = "1.97" [workspace.dependencies] -cellule-app = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } -cellule-host = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } -cellule-peer-http = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } -cellule-runtime = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } -cellule-store = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } -cellule-ltx = { git = "https://github.com/crabbuild/cellule.git", rev = "9e17746a633ca1046bd93866866074226091f81a" } +cellule-app = { git = "https://github.com/crabbuild/cellule.git", rev = "1d0648b3b5ae0cca040c614b505b4d72bb769ab1" } +cellule-host = { git = "https://github.com/crabbuild/cellule.git", rev = "1d0648b3b5ae0cca040c614b505b4d72bb769ab1" } +cellule-peer-http = { git = "https://github.com/crabbuild/cellule.git", rev = "1d0648b3b5ae0cca040c614b505b4d72bb769ab1" } +cellule-runtime = { git = "https://github.com/crabbuild/cellule.git", rev = "1d0648b3b5ae0cca040c614b505b4d72bb769ab1" } +cellule-store = { git = "https://github.com/crabbuild/cellule.git", rev = "1d0648b3b5ae0cca040c614b505b4d72bb769ab1" } +cellule-ltx = { git = "https://github.com/crabbuild/cellule.git", rev = "1d0648b3b5ae0cca040c614b505b4d72bb769ab1" } async-trait = "0.1" blake3 = "1.8" futures-util = "0.3" @@ -61,6 +61,7 @@ extenddb-storage = { git = "https://github.com/crabbuild/extenddb.git", rev = "7 futures-util.workspace = true fs4 = "0.13.1" getrandom = "0.3" +object_store.workspace = true reqwest.workspace = true serde.workspace = true serde_json.workspace = true @@ -74,12 +75,12 @@ uuid.workspace = true zeroize = "1" [dev-dependencies] +prost = "=0.13.5" cellule-store = { workspace = true, features = ["test-support"] } aws-config.workspace = true aws-credential-types.workspace = true aws-sdk-dynamodb.workspace = true ed25519-dalek.workspace = true extenddb-engine = { git = "https://github.com/crabbuild/extenddb.git", rev = "7eaa89b437feed0af0f05883d3f1493f86c6fc6d" } -object_store.workspace = true tempfile.workspace = true tokio = { workspace = true, features = ["macros", "rt-multi-thread", "net", "time"] } diff --git a/README.md b/README.md index 4441d43..1a6b1cb 100644 --- a/README.md +++ b/README.md @@ -55,6 +55,10 @@ partition-key ranges. A GSI is maintained asynchronously from a journal committed with the base item. Cross-Cell transactions use a durable coordinator decision and idempotent participant resolution. +### Choose placement per table + +Set the `CreateTable` tag `beyonddb:cell-model` to `single`, `auto`, or `partitioned`. Keep one base data Cell for a table that fits its resource budgets, start at one and permit growth, or provision multiple ranges immediately. Omitting the tag preserves `initial_partitions`. GSIs use separate Cells; changing the tag later does not migrate the table. Live model conversion is not implemented. See [Cell model selection](docs/scaling.md#choose-a-tables-cell-model) and the [AWS CLI example](docs/user-guide.md#create-a-table-and-wait-for-it). + ### When a write becomes durable ![Sequence of a signed PutItem: ExtendDB validates, BeyondDB routes, Cellule commits and publishes LTX, then the response returns](diagram/beyonddb-architecture/durable-write.svg) @@ -64,11 +68,13 @@ stream intent commit in one Cell command. The successful response follows durable publication. [PNG version](diagram/beyonddb-architecture/durable-write@2x.png) · [Detailed architecture and recovery design](docs/architecture.md) -The serving binary currently waits for object-store publication on each -durable write. Cellule's follower-log mode is not enabled in BeyondDB. An -opt-in persistent follower store and authenticated peer transport exist, but -the durability provider is not installed in serving and successor recovery -is unfinished. See the [follower durability design](docs/follower-durability.md). +The default serving path waits for object-store publication on each durable +write. Experimental `follower_durability_enabled` installs Cellule's node-log +provider and can acknowledge after every enrolled follower fsyncs the commit. +A three-process signed SDK test verifies item mutations, same-partition +transaction replay, and stream records after an owner kill with object uploads +withheld. Broader fault and performance qualification remains open. See the +[follower durability guide](docs/follower-durability.md). ## Current capability boundary diff --git a/docs/api.md b/docs/api.md index 94d3ccc..e542fc8 100644 --- a/docs/api.md +++ b/docs/api.md @@ -28,6 +28,12 @@ The [implementation record](implementation-status.md#current-verified-slice) maps these paths to tests. The full upstream protocol suite remains an [acceptance gate](implementation-status.md#acceptance-proof-for-a-server-claim). +### Creation-time Cell model extension + +`CreateTable.Tags` accepts `beyonddb:cell-model` with `single`, `auto`, or `partitioned`. `single` fixes the base table to one data Cell; `auto` starts at one and permits splitting; `partitioned` starts at the configured count, with a minimum of two. Omitting the tag retains existing behavior. The choice is durable table-generation metadata, not an AWS DynamoDB setting. + +Later tag operations change metadata only. They do not change placement, and `UpdateTable` does not convert models. GSIs stay separate; LSIs stay with their base items. Read [Cell model limits](scaling.md#choose-a-tables-cell-model) and the [CLI creation example](user-guide.md#create-a-table-and-wait-for-it). + ## Indexes Local secondary indexes (LSIs) share a base Cell's atomic mutation. Global secondary indexes (GSIs) have separate owner Cells and receive base changes asynchronously. diff --git a/docs/architecture.md b/docs/architecture.md index 8e2738f..bac7ad3 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -95,7 +95,7 @@ sequenceDiagram Driver->>B: PREPARE item images and locks B-->>Driver: Durable prepare receipt end - Driver->>Coord: Publish COMMIT or ABORT + Driver->>Coord: Record prepares + COMMIT, or publish ABORT Coord-->>Driver: Durable decision par Resolve owners Driver->>A: Apply or discard decision @@ -108,7 +108,7 @@ sequenceDiagram Driver-->>Client: Return or replay result ``` -The original coordinator and participant identities survive routing changes. If a driver dies after publishing the decision, startup or serving recovery finishes participant resolution. A caller timeout leaves the outcome unknown until the durable decision is inspected. A client token distinguishes a matching replay from a different request. Transactional reads use shared key locks and captured images. See the [transaction protocol and failure cases](cross-cell-transactions.md). +The original coordinator and participant identities survive routing changes. If a driver dies after publishing the decision, startup or serving recovery finishes participant resolution. A caller timeout leaves the outcome unknown until the durable decision is inspected. A client token distinguishes a matching replay from a different request. Cross-Cell transaction reads use shared key locks and captured images. Single-Cell reads use one atomic query with a compact internal response when it fits the query envelope. See the [transaction protocol and failure cases](cross-cell-transactions.md). ## Range split and GSI projection diff --git a/docs/cross-cell-transactions.md b/docs/cross-cell-transactions.md index 4179dbc..a411b62 100644 --- a/docs/cross-cell-transactions.md +++ b/docs/cross-cell-transactions.md @@ -63,15 +63,33 @@ Driver or recovery worker: idempotent participant resolution | After prepare | Durable action | Caller-visible result | | --- | --- | --- | -| Every participant prepared | Coordinator changes `BEGIN` to `COMMIT`. | Success follows durable apply and a resolution receipt from every participant. | +| Every participant prepared | One coordinator command records prepare receipts and changes `BEGIN` to `COMMIT`. | Success follows durable apply and a resolution receipt from every participant. | | A condition definitively fails while still in `BEGIN` | Coordinator changes `BEGIN` to `ABORT`. | Return the ordered cancellation result after abort resolution. | | A reply or caller times out | Read the coordinator decision and participant state during retry or recovery. | The timeout alone does not establish `COMMIT` or `ABORT`. | `COMMIT` and `ABORT` are immutable. A participant that has not yet applied a durable `COMMIT` remains locked until resolution. -Two keys in the same Cell share one prepare and one resolution. The public -adapter uses this protocol even when every key belongs to one Cell. +The normal successful path uses `CommitPreparedTransaction` to publish the +prepare receipts and COMMIT together. It still verifies that every participant +has durable prepare evidence; an incomplete or mismatched receipt set cannot +commit. Existing phase commands remain available for recovery and older driver +paths. A lost response requires reading the authoritative decision before +resolving participants. The [driver regression](../tests/elastic_cells/transaction_driver.rs) +and [commit assertions](../tests/elastic_cells/transaction_commit.rs) cover +partial/wrong receipts, one coordinator commit, replay, coordinator restoration, +competing drivers, and lost replies. + +### One Cell and multiple Cell requests + +A table created with `beyonddb:cell-model=single` keeps its base items in one dedicated Cell. Small transaction reads whose operations all resolve to that Cell use one atomic query. Cross-table requests and reads requiring saved images can still need coordination. GSIs remain asynchronous and separate. + +Two write keys in the same Cell share one prepare and one resolution. Without a +client token, a write whose operations all route to one Cell uses one atomic +local command. Requests with a client token retain the durable coordinator. A +single-Cell `TransactGetItems` instead uses one atomic Cell query with a compact +internal response; outputs that exceed its envelope use the durable saved-image +fallback. ### Atomicity includes what concurrent readers can observe @@ -82,7 +100,7 @@ return a retryable error. It must not return B's old image. The lock check and item read occur inside one Cell query, including when the item does not yet exist. The coordinator's durable COMMIT is the logical write serialization point. -`TransactGetItems` prepares shared locks and immutable read images in every +Cross-Cell `TransactGetItems` prepares shared locks and immutable read images in every participant. Once the last shared prepare succeeds, all captured keys remain locked and those images coexist; this supplies its serialization point. Read resolution releases locks but preserves the saved images for response assembly. @@ -146,7 +164,7 @@ decision evidence tied to the transaction, participant set, and fenced authority | Safety qualification | Deterministic failure tests cover selected schedules | Exercise concurrent transfers, conditional write skew, read transactions, owner replacement, and lost replies with a recorded-history checker. Check conservation, serializability, and no duplicate effects at every injected phase cut. | | Bounded history | Completed decisions, participant markers, and committed read images are retained | Design an acknowledged retirement boundary that rejects late phase messages before deleting tombstones; include split/backup pins and outstanding read fetches. Ten-minute client token expiry alone cannot authorize deletion. Prove storage reaches a steady state under a soak workload. | | Admission | A 100-operation participant reserves about 278 MiB at 4-KiB pages before index/journal claims and payload allowance | Qualify the allocation bound and contention cost; add WAL, disk, and heap admission. Coordinator progress also needs capacity to record decisions and receipts. Never reclaim an unresolved participant's claim on timeout. | -| Latency | Sequential prepare; up to four terminal resolutions in flight per call; at least `5P + 3` durable commands for single-chunk inputs | Measure publication and RPC time by participant count. Evaluate prepare parallelism and batched coordinator progress with renewed crash/concurrency proof before changing those phases. | +| Latency | Up to eight prepares and four terminal resolutions overlap. Prepare receipts and COMMIT share one coordinator command; provisioning, payload uploads, resolution receipts, and read-result cleanup add work. | Measure cold admission separately from warm phase execution, by participant count. The combined decision removes one publication; end-to-end SQLite parity still requires measurement and further work. | | Fleet recovery | 4,096 fixed coordinator shards per account; the worker selects one shard per 250-ms tick | Integrate placement and recovery scheduling with bounded concurrency and backlog metrics. A nominal pass over 4,096 known shards already takes about 17 minutes before slow work; this is arithmetic, not measured RTO. | | Data distribution | HASH-key siblings share one Cell with a finite database budget | Qualify skew, hot keys, split headroom, and oversized item collections. More Cells do not distribute one key's lock or split a single HASH group in the current layout. | | API completion | ALL-projection LSIs share participant resolution; initial GSIs use a durable asynchronous journal; committed writes append stream records locally. Non-ALL LSIs, online GSI lifecycle, Streams policy transitions, and aggregate evaluated Update-size semantics remain incomplete or unqualified. | Complete the engine read contract, index lifecycle, and Streams qualification; compare size and error semantics with AWS before claiming compatibility. | @@ -204,15 +222,64 @@ finish a terminal decision across account and data Cell participants, using part state after an ambiguous reply and recording each resolution durably. A transaction driver can also resume a published `BEGIN`: it reads the -immutable participant payloads, prepares in Cell order, records receipts, +immutable participant payloads, prepares with bounded concurrency, records receipts, publishes one decision, and finishes resolution before returning that decision. +For small transactions, one coordinator query observes the durable status and +all unprepared participant payloads together. This replaces separate status, +participant-list and payload requests. It accepts at most 32 KiB of saved +payload bytes and keeps the complete encoded reply within 64 KiB. Large or +multi-chunk inputs use the existing chunk protocol. Prepared participants need +no payload reload; terminal decisions return status only. + +```text +Coordinator query: status + small immutable payloads + | + v +Participant prepares (up to eight concurrently) + | + v +Durable prepare evidence + COMMIT + | + v +Participant resolution + final coordinator status +``` + +The combined query changes preparation discovery only. It retains published +BEGIN, immutable targets, idempotent participant commands, durable decision and +resolution checks. The compiled coordinator code and query contract include +this operation; peers must use the matching compiled release. Stored schemas +and payload formats are unchanged. + Concurrent resumes and lost prepare/decision replies use durable state as the authority. Definitive condition, lock, or routing failures request an abort; an already-published terminal decision wins. Transport uncertainty leaves recoverable work and never becomes cancellation. Shard admission now registers a fixed shard number in the account Cell before it returns to a caller. +Fresh coordinator Cells can bootstrap concurrently when residency has spare +slots. A fixed set of 64 local gates serializes the same Cell; collisions can +also delay independent Cells. Each bootstrap holds shared admission and a +pending-slot reservation until Cellule accounts for activation or the caller +finishes. Cellule's active-slot count includes in-flight activations, so a +cancelled caller does not make a still-running activation disappear from +capacity accounting. Published-owner restore, transfer, and reclamation retain +exclusive admission. If no spare slot can be reserved, the caller takes that +exclusive path before claiming authority. + +```text +Coordinator A: reserve -> authority CAS -> bootstrap + publish -> register A -> BEGIN A +Coordinator B: reserve -> authority CAS -> bootstrap + publish -> register B -> BEGIN B + shared bounded slot accounting; independent requests can overlap +``` + +Each request waits for its own published coordinator root and durable account +registration before BEGIN. Concurrency does not change lease fences, authority +CAS, initial publication, participant prepare, or coordinator decision rules. +The [SDK regression](../tests/peer_network/residency/coordinator_admission.rs) +checks overlapping distinct Cells, serialized token replay, last-slot admission, +cancellation, durable registration, and restoration. + On startup, the server pages that account-owned registry, recovers idle shards and shards whose owner lease expired, including when the replacement uses a different endpoint, then aborts unfinished @@ -932,6 +999,16 @@ after fenced recovery resolves every abort. SQL query plans use the token and transaction-ID indexes rather than scanning coordinator history. +### Repeated admission of resident coordinators + +After durable registration, the provisioner retains a bounded in-memory receipt. +It skips repeated registration and authority lookups only while the account and +coordinator are both resident with the same incarnations and compiled code/schema. +A registration learned through a query must be covered by the published account +root before it can populate this cache. Drain, remote ownership, a missing receipt, +or a new provisioner uses canonical admission. The shortcut does not change +BEGIN, durable decisions, participant resolution, or dispatch fencing. + ## Serving-time recovery `CellInitialPartitionProvisioner::install_transaction_recovery_loop` installs @@ -946,6 +1023,11 @@ account's cursor advances before owner activation and wraps to find later registrations. Live remote owners stay in place. Idle or expired owners use the same catalog validation, node fencing, and Cell authority CAS as startup. +Discovery first restores the account Cell that stores the shard registry if its +owner is Idle or expired. Otherwise, losing that owner would prevent the worker +from learning which new coordinator shards need recovery. A live account owner +continues to serve the registry query; takeover still requires the normal fence. + A cached empty-work receipt skips an Idle shard only when its incarnation and published commit sequence match. Unknown or changed roots must be inspected; completed history must not continuously churn the active-Cell pool. This cache diff --git a/docs/deployment.md b/docs/deployment.md index 1168d3f..4090049 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -63,6 +63,13 @@ Wait for RustFS to accept S3 requests before creating the bucket. Keep its named volume for restart testing; removing it removes this fixture's data. Use a service with qualified conditional writes for any nonlocal deployment. +On macOS with Colima, this named volume keeps provider data inside the VM. +A direct S3 diagnostic +measured faster 1-KiB conditional writes on native volumes than host file +sharing in both repetitions. GET results and host load differed; this is a +local fixture observation, not a production capacity guarantee. Record the +storage mount when comparing performance. + ## Prepare keys, policy, and configuration The example uses `/etc/beyonddb` for secrets and `/srv/beyonddb` for scratch files. Run the provisioning commands with an account that can write those paths, and run the server with read access to the key and certificate files. For a local account without that access, change every path in the commands and JSON to directories it owns. @@ -161,21 +168,38 @@ metadata immediately; changes made through another node become visible after the cache TTL. Enabling the flag also caches complete routed directory pages for point operations. The owning data Cell rejects a stale epoch and the server drops that route entry, so a split is refreshed on the next request. It -also keeps immutable catalog proofs and resident local Cell handles for 500 ms; -the authority check resumes after that window, and a drained Cell handle still -rejects work immediately. This short owner cache improves warm local latency -while bounding visibility of an ownership change. +also caches immutable catalog proofs. Resident Cell handles are cached for +500 ms on both public request resolution and private peer invocation. After +expiry, the runtime resolves a still-resident actor without reading object +storage. Cached handles are checked against the current resident actor before +reuse, and dispatch fences drained handles. Remote routing, local admission +and recovery still use exact authority checks. Peer enrollment and +authorization are checked for every private request. -The parser rejects unknown fields. `initial_partitions` defaults to one and can provision 1–256 initial data Cells per new table. `max_active_cells` defaults to 128 and reserves the Cell runtime capacity for account, coordinator, management, and data Cells together; size it for the number of simultaneously resident Cells on the node and the available memory. `sql_workers` is optional; when omitted, the runtime derives the worker count from host parallelism, capped at sixteen. Set it explicitly when a node serves many partitions and you have measured enough CPU and memory headroom. Each worker owns its SQLite connections, so increasing the value does not make one hot Cell publish concurrently. The split threshold defaults to 256 MiB of occupied SQLite pages. `node_id` identifies a physical node; each running node needs a distinct ID and scratch path. +The parser rejects unknown fields. `initial_partitions` defaults to one and accepts powers of two from 1–256. It sets the initial base and GSI range count when a new table omits the placement selector. A creation-time `beyonddb:cell-model` tag overrides that count: `single` and `auto` start with one; `partitioned` uses at least two. The model is stored with the table generation, so later node configuration changes do not repartition existing tables. See [choose a table's Cell model](scaling.md#choose-a-tables-cell-model). `max_active_cells` defaults to 128 and reserves the Cell runtime capacity for account, coordinator, management, and data Cells together; size it for the number of simultaneously resident Cells on the node and the available memory. `sql_workers` is optional; when omitted, the runtime derives the worker count from host parallelism, capped at sixteen. Set it explicitly when a node serves many partitions and you have measured enough CPU and memory headroom. Each worker owns its SQLite connections, so increasing the value does not make one hot Cell publish concurrently. The split threshold defaults to 256 MiB of occupied SQLite pages. `node_id` identifies a physical node; each running node needs a distinct ID and scratch path. `follower_store_bytes` is an optional positive disk budget for persistent -follower lanes under `data_dir/follower-store`. Setting it opens a private, -authenticated node-log receiver; it does **not** enable follower durability or -improve write latency yet. BeyondDB still waits for object-store publication -and advertises no follower capacity. Reserve this budget in addition to -`disk_budget_bytes`, and retain the follower directory across process restart. -The [follower durability guide](follower-durability.md) tracks the remaining -enrollment, lifecycle, and recovery work. +follower lanes under `data_dir/follower-store`. Reserve it in addition to +`disk_budget_bytes`, and retain the directory across process restart. By +itself, it opens the authenticated receiver for recovery and advertises zero +follower capacity; writes still wait for object publication. + +Experimental `follower_durability_enabled: true` requires a follower-store +budget of at least 64 MiB. It installs the host provider during startup, gates +recruitment until configured-account recovery completes, and advertises actual +remaining follower bytes only after the private listener starts. Each enrolled +member must fsync before a follower proof; absent an eligible ensemble, the +object-publication path remains available. Use separate node IDs, certificates, +and data directories for every server. The three-process crash test exercises +same-partition item mutations, transaction replay, and stream records; +independent-host failure, cross-partition transaction faults, rotation under +load, and sustained throughput still need qualification. See the +[follower durability guide](follower-durability.md). + +The S3 serving path uses Cellule's provider builder, including multipart +conditional copy support needed to pin recovery overlays. The URL-only builder +previously left that operation unconfigured, which the real RustFS process-kill +test exposed. GCS and Azure retain their URL-based configuration paths. | Credential or file | Used by | Keep across restart? | | --- | --- | --- | diff --git a/docs/follower-durability.md b/docs/follower-durability.md index bf63e03..6031a7e 100644 --- a/docs/follower-durability.md +++ b/docs/follower-durability.md @@ -2,7 +2,7 @@ ## Why this work is needed -The current BeyondDB server acknowledges a mutation after its Cell state is +The default BeyondDB server acknowledges a mutation after its Cell state is published to the configured object store. On the recent four-partition local RustFS fixture, this path remained far below file-backed ExtendDB SQLite for `PutItem`, `UpdateItem`, `BatchWriteItem`, and transactions. A separate @@ -14,13 +14,14 @@ amounts of item SQL cannot eliminate the object publication round trip. The pinned Cellule revision already exposes a node-log durability supervisor, follower stores, and a durability gate. BeyondDB has a lease-bound authority adapter, an opt-in inbound follower receiver, a pinned mTLS outbound -transport, and a host-provider enrollment adapter. The provider is not -installed in the serving binary, and owner recovery is incomplete. No -follower mode should be advertised or enabled until that integration is -proven. +transport, and a host-provider enrollment adapter. Experimental +`follower_durability_enabled` now installs that provider. Recruitment waits for +startup recovery; available follower capacity is advertised only after the +private receiver starts. A three-process signed SDK crash test passes, while +broader fault and performance qualification remains open. ```text -Current serving path Proposed multi-node path +Default serving path Experimental multi-node path AWS SDK request AWS SDK request | | @@ -49,7 +50,7 @@ fallback. These are durability conditions, not optional performance hints. | --- | --- | | Follower store | Open `FollowerStore` in each node's durable data directory, reserve disk, and retain lanes across process restart. Advertise follower capacity only after the store and authenticated listener are ready. | | Peer transport | Implement Cellule's `NodeLogTransport` append, seal, retire, and bounded tail operations over the private mTLS listener. Pin each remote certificate to its live node advertisement. Bound request bytes, time, and concurrent work. | -| Follower authorization | Match the mTLS identity to the advertised session. Use `NodeDirectory::authorize_log_append`, `authorize_log_retire`, and `authorize_log_recovery` before touching a lane. Reject wrong members, epochs, coverage watermarks, and unfenced recovery attempts. | +| Follower authorization | Match the mTLS identity to the advertised session. For each append, load one fresh `NodeDirectory::peer_verifier`, match caller to leader, and consume `EnrolledPeerVerifier::authorize_log_append`. Retirement and recovery use `NodeDirectory::authorize_log_retire` and `authorize_log_recovery` before touching a lane. Reject wrong members, epochs, coverage watermarks, and unfenced recovery attempts. | | Enrollment and authority | Implement `NodeDurabilityProvider` using `NodeDirectory::try_recruit_log`. Implement `NodeLogAuthority` with the directory's activate, coverage, and close CAS operations; reconcile CAS races with lease heartbeats without losing the enrolled log. Supply the exact session, node ID, members, lease guard, transport, and limits to `NodeDurabilityConfig`. | | Recovery | Before public readiness after an owner loss, seal and fetch the failed owner's authorized follower tail, reconcile object coverage, and restore acknowledged Cell commits. Do not return success for a write whose proof cannot be recovered on a successor. | | Lifecycle | Rotate and retire only after the recorded coverage barrier. Drain the node, settle publications, and preserve follower files if withdrawal or recovery has not completed. | @@ -58,7 +59,13 @@ The private `BeyonddbPeers` router now has a bounded node-log receiver when `follower_store_bytes` is configured. It opens Cellule's persistent `FollowerStore` beneath `data_dir`, requires a live mTLS identity bound to the advertised session, and checks the directory's append, retirement, or fenced -recovery authority before touching a lane. The outbound +recovery authority before touching a lane. Append identity and authority use the same fresh canonical signed record: +certificate/key/fleet/image/release are validated during enrollment, then lease +expiry is rechecked after I/O along with ensemble, epoch, open phase, and covered +watermark. The canonical read is the authorization observation for this request; +concurrent record changes after it do not cause a second read. A new proof is +loaded for every request. Local lease fencing, durable append and the final +response fence still apply. The outbound `PeerNodeLogTransport` resolves a live advertised member, pins its certificate and key, and bounds requests, replies, and tail paging. Tests cover a durable append and duplicate append over real two-identity mTLS, follower-store @@ -71,17 +78,19 @@ persisted tail through the bounded witness reader; sealing before the claim is rejected. The test also confirms that a recovery claimant must advertise Cellule's node-log protocol even when it offers no follower bytes. With an opt-in persistent store, the serving binary now advertises that protocol but -zero follower bytes, so it can claim recovery without being recruited for -write acknowledgments. Retirement and successor Cell overlay attachment -still need end-to-end tests. +zero follower bytes unless experimental follower durability is also enabled. +It can therefore claim recovery without being recruited for write +acknowledgments. The process test now exercises successor overlay attachment; +retirement under concurrent load still needs qualification. The lease-bound `PublishedNodeLogAuthority` adapter can enroll a follower set and apply the directory's activation, coverage, and close transitions. It serializes those mutations with heartbeat refreshes and reloads the exact session after an ambiguous CAS. `PeerNodeDurabilityProvider` gives Cellule's host supervisor the enrolled members, authority, transport, and lease for -each epoch, but the serving binary does not install it while successor -recovery is unfinished. It advertises no usable follower capacity. Adding a +each epoch. The server installs it before `CellNode::start` and uses a startup +gate to keep recruitment disabled until recovery finishes. With the option +enabled, ready receivers advertise their remaining disk budget. Adding a local in-process follower under a second logical node ID would not provide an independent failure domain and must not be used as a production durability shortcut. @@ -95,9 +104,54 @@ Cell takeover encounters an active, untiered node log, the opt-in serving path claims fenced recovery, runs this coordinator, and only then passes its takeover proof to Cellule. A product test now captures a real account Cell frame, appends it to a persistent follower lane, and verifies that a -successor restores the untiered commit before takeover. This path still -lacks a multi-node crash-after-acknowledged-write SDK test; the provider -remains uninstalled and follower-backed acknowledgments remain disabled. +successor restores the untiered commit before takeover. A separate +[server process test](../tests/server_binary/follower_durability.rs) now kills +the owner after SDK success while its data Cell object uploads are withheld. +The replacement recovers PutItem, UpdateItem return values, DeleteItem, +BatchWriteItem, and a same-partition TransactWriteItems outcome. Replaying the +transaction preserves its conditional-insert result, and the AWS Streams CLI +finds exactly one corresponding record for each tested mutation. All nodes use +separate processes, keys, and persistent directories on one workstation. + +This test also exposed an S3 configuration gap: recovery's immutable overlay +pinning requires conditional copy support. BeyondDB now uses Cellule's S3 +provider builder, which configures multipart conditional copies. No dependency +source was patched. + +The signed SDK component test in +[`tests/peer_network/residency/follower_durability.rs`](../tests/peer_network/residency/follower_durability.rs) +now connects the real host provider to two persistent followers over mTLS. +It withholds only the data Cell's immutable object uploads and verifies that +the signed `PutItem` succeeds while the object root remains unchanged. After +fencing the owner and stopping renewal, a successor recovers the item through +the authorized follower tail. A matching object-only control uses the same +fixture and confirms that the write waits for publication. + +This test runs the nodes within one process. It establishes the composition +of enrollment, fsync proof, SDK acknowledgement, and successor recovery; +it does not by itself establish process-crash durability, independent failure +domains, transaction or stream recovery, or release throughput. The separate +server test above covers the listed process-crash cases. The provider must be +installed during node startup, before `CellNode::start`. Cellule selects a +complete follower ensemble from the live fleet, so this fixture's additional +frontend means two eligible followers are required for recruitment. + +## Enable the experimental path + +Add these fields to each node's existing server configuration: + +```json +{ + "follower_store_bytes": 1073741824, + "follower_durability_enabled": true +} +``` + +Use distinct node IDs, certificates, and persistent directories. A three-node +fixture provides two enrolled followers for each leader. When an eligible +ensemble is unavailable, SDK writes retain the object-publication path. The +server verification and release comparison +records the tested scope, request errors, and remaining performance gaps. ## Verification before comparing throughput diff --git a/docs/implementation-status.md b/docs/implementation-status.md index 04027cc..59cb609 100644 --- a/docs/implementation-status.md +++ b/docs/implementation-status.md @@ -158,6 +158,21 @@ starts a digest-pinned RustFS 1.0 GA container with an isolated Docker volume and a random loopback port; no native RustFS installation is used. Colima users can select their daemon with `DOCKER_CONTEXT=colima`. +When Docker's own volume filesystem is short of space or inodes, select a host +bind root that the daemon shares. Each test creates a fresh child directory +there. For example, with Colima's home-directory sharing: + +```bash +export BEYONDDB_TEST_RUSTFS_BIND_ROOT="$HOME/.codex/tmp/beyonddb-rustfs-tests" +CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/beyonddb-sdk-verification" \ + cargo test --locked --test server_binary -- --ignored --test-threads=1 +``` + +Successful tests remove their owned bind data; failed tests retain it and print +the path. A bind mount still requires enough free Docker VM inodes to create a +container. Preserve failed fixture evidence before cleaning up owned test +resources. + ```bash CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/beyonddb-cellule \ cargo test -p beyonddb --test server_binary -- --ignored --test-threads=1 diff --git a/docs/performance.md b/docs/performance.md index 94551ef..f644e07 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -1,25 +1,971 @@ # Measure BeyondDB performance -BeyondDB does not yet have a qualified production throughput or latency target. Earlier September 2026 single-node samples suggested that warm point reads could exceed the file-backed SQLite fixture, while durable writes and transactions remained slower. The refresh below did not reproduce those high read rates. Do not use these numbers to plan a fleet. BeyondDB's request path and durability contract differ: an item write waits for Cellule to publish the committed Cell state to an object store. - -The latest full signed-API rerun pinned Cellule to `9e17746` and ExtendDB -SQLite to `7eaa89b`. Both completed all 24 five-second cases without -foreground SDK errors. At eight clients, BeyondDB/SQLite measured 235/507 -`GetItem`, 78/55 `PutItem`, and 1.50/131 `TransactWriteItems` requests/s. -BeyondDB exceeded SQLite in BatchGetItem and metadata reads at both client -counts, and in PutItem at eight clients, but did not meet the all-API target. -It logged deferred background work. External host load changed materially -between fixtures. See the [full table, p95 latencies, fixture, and raw -JSON](../benchmarks/2026-09-29-cellule-main-rerun/README.md); the result does -not establish production capacity or a controlled speed ratio. - -An [earlier attempt with the same pin](../benchmarks/2026-09-29-cellule-main-attempt/README.md) +BeyondDB does not yet have a qualified production throughput or latency target. Earlier September 2026 single-node samples suggested that warm point reads could exceed the file-backed SQLite fixture, while durable writes and transactions remained slower. The refresh below did not reproduce those high read rates. Do not use these numbers to plan a fleet. BeyondDB's request path and durability contract differ: by default an item write waits for Cellule to publish committed state to object storage. Experimental follower durability can acknowledge a durable follower receipt while object tiering continues. + +Benchmark reports, raw logs and fixture snapshots are kept locally and excluded from Git. The summaries below retain measured revisions and limitations; durable behavior is checked by the committed integration tests. + +## Active owner lookup during restore + +The local handle cache validates the exact owner fence, code and schema against Cellule's active-owner capability. Sparse and hydrating owners can serve foreground work while background hydration continues. An expired cache entry can be rebuilt from that capability without an authority read. A miss follows the normal catalog and fresh-authority path; it does not authorize acquisition. Peer enrollment and request authorization are still checked for every invocation. + +```text +Signed peer request + │ + ▼ +Enrollment + authorization + │ + ▼ +Active owner lookup ── found ──► Check fence / code / schema ──► Dispatch + │ │ + absent Actor admission rechecks + │ lease, drain and ownership + ▼ +Catalog + fresh authority + │ + ▼ +Normal resolution / eligible recovery +``` + +A controlled restored-owner regression holds background hydration open while five signed foreground reads run, including one after cache expiry. The earlier resolver performs five receiver authority reads; the active-owner path performs zero. A separate owner-reacquisition regression rejects reuse of a cached handle from an earlier epoch of the same incarnation. These checks prove reduced metadata work and correct cache invalidation. They do not establish an end-to-end throughput improvement or resolve the remaining write/transaction durability costs. + +## Recovery verification and owner placement + +A range can release residency and publish an Idle root while its former node is still alive. Later recovery may acquire that root on another node. Recovery checks therefore compare the retained Cell ID, incarnation, code, schema, owner epoch and published commit sequence, followed by signed SDK item reads. A fixed owner name alone does not establish durable recovery. + +A focused transaction regression installs discovery before coordinator registration, leaves one participant completion receipt unrecorded, and verifies that recovery preserves the exact live-owner fence. After the owner's lease expires, the serving worker must record both participant resolutions before signed SDK reads confirm the recovered values. + +Authentication also restores a cataloged credential shard after its owner expires. This uses the existing fenced takeover path and requires a published root and fresh lease evidence. The regression covers enabled and disabled peer caches, refusal to replace a live owner, and an unknown access key that creates neither a catalog entry nor authority. + +The reviewed Cellule EOF fix keeps observed object streams finished after completion. Both new regressions reproduce the original panic before the fix. With the updated pin, the original follower durability process test passes, including withheld object publication, owner process kill, recovery, and graceful shutdown. This closes that local failure; full fleet recovery and sustained performance still require qualification. + +## Why writes and transactions cost more than SQLite + +The pinned ExtendDB SQLite backend uses WAL with `synchronous=NORMAL`. SQLite documents that this mode does not synchronize the WAL after every commit; `FULL` adds a synchronization for each commit. BeyondDB waits for object-store publication or a durable follower receipt before acknowledging a mutation. The benchmark therefore compares different durability paths. [SQLite durability reference](https://www.sqlite.org/pragma.html#pragma_synchronous) + +```text +ExtendDB SQLite BeyondDB follower mode +request request + | | +SQLite transaction owning Cell command + | | +WAL commit follower append + fsync + | | +response durable receipt -> response + | + async object tiering +``` + +The latest completed release comparison measured SQL commands at 0.800 ms on average in four-Cell mode and 1.059 ms in single-Cell mode. Command responses with follower durability averaged 101.332 and 85.691 ms respectively. These populations include background work and overlap other scopes; they are evidence of costs around SQL, not an additive explanation of request latency. The host was heavily loaded and using about 36–37 GiB of swap, which limits comparison with earlier samples. + +Cross-Cell transactions add a durable coordinator admission, preparation, decision, participant resolution, and cleanup. A fresh coordinator shard can also require authority and catalog I/O. Boto3 automatically supplies an idempotency token for `TransactWriteItems`; BeyondDB preserves account-scoped replay and mismatch checks through the coordinator even when all items occupy one Cell. + +[Single-Cell placement](scaling.md#choose-a-tables-cell-model) removes cross-Cell work from eligible transaction reads and reduces the number of write participants. It retains follower durability and tokenized write coordination. Reaching SQLite write latency requires further work on durable I/O, batching, and coordinator admission, plus measurements with declared durability settings. It is not an established property of the new placement option. + +### Where optimization can help + +| Cost | What can reduce it | What must remain correct | +| --- | --- | --- | +| One durable publication per independent mutation | Coalesce mutations targeting the same Cell. | Individual conditions, results, stream records and durable acknowledgements. | +| Transaction prepare publications | Coalesce independent participant prepares targeting the same Cell. | Separate transaction identities, locks and coordinator decisions; atomic rollback of a rejected batch. | +| Fresh coordinator admission | Reuse admitted owners and coalesce discovery/catalog work. | Account-scoped token replay and mismatch checks, owner fencing and recovery discovery. | +| Several participants for a small table | Start with one data Cell using `single` or `auto`. | Disk and write budgets, index placement and recovery time. | +| Follower/object I/O | Measure and reduce transport, authority lookup and storage latency. | The chosen durability contract and restart survival. | + +These changes address different parts of the request. A single data Cell can remove participant fan-out, but tokenized transaction writes still have several serial durability barriers. Lower SQL execution time alone cannot remove those barriers. + +### Avoid stale heartbeat writes after local log updates + +A completed node-log update changes the node advertisement's ETag. Previously, the heartbeat retained its older version, so its next renewal first attempted a stale conditional write, then read the current record and retried. The publisher and its log adapter now share their newest completed canonical observation. Late replies cannot replace a higher generation. + +```text +Local log update -> completed record + ETag -> shared version hint + | +Heartbeat -------------------------------------------+ + | + conditional write + | + conflict -> canonical read -> retry +``` + +The hint reduces the measured local-update renewal path from **two PUTs and one GET to one PUT**. It grants no authority: the directory still checks the signed identity, transition and exact ETag. Log operations continue to read canonical state; unseen updates still require a conflict and authoritative rebase. No provider I/O holds the shared observation lock, and the existing 15-second lease and fencing checks remain. + +The [heartbeat regression](../tests/node_log_authority/heartbeat_versions.rs) injects 6.5-second conditional-write latency after a completed log coverage update. The old implementation fences before its first renewal; the updated implementation completes two successive renewals and remains live. Delayed read replies, unseen changes and stalled log I/O are also checked. This isolates an avoidable renewal cost; it does not establish that every previous benchmark fence has this cause or prove SQLite throughput parity. + +### Share transaction prepare publications + +Independent small transactions targeting the same data Cell can share a prepare publication. Each transaction retains its own intent, locks and coordinator decision. The queue is bounded to 64 pending prepares per Cell; each batch contains at most 16 prepares and 32 KiB of input, with a 2 ms collection window. A single selected prepare uses the existing command. Legacy account participants and large payloads retain their existing paths. + +```text +Transaction A: durable BEGIN -> prepare A --+ + | one data Cell publication +Transaction B: durable BEGIN -> prepare B --+ stores separate intents and locks + | + durable receipt for each coordinator + / \ + A: COMMIT/ABORT B: COMMIT/ABORT + | | + resolve A resolve B +``` + +A rejected prepare rolls back the whole batch. Individual commands follow only a durable rejected receipt or a proven refusal before the batch starts. Uncertain replies trigger durable participant-state reads. Coordinator decisions, participant resolutions and stream records retain their existing semantics. + +The signed SDK verification reduces **64 durable data commands to 38–40** for 32 independent two-item transaction writes. It checks complete results after owner restoration and recovery after losing a batch reply following durable publication. **43 distinct tests, formatting and strict Clippy pass.** Batch grouping varies with scheduling. This proves reduced durability work; release throughput and SQLite parity require separate measurement. + +### Avoid repeating acknowledged transaction work + +For a fresh read transaction involving only routed data Cells whose BEGIN payload fits 32 KiB, an acknowledged BEGIN fixes the participant list and its ordering. BeyondDB retains that list while the existing prepare, COMMIT, and resolution path completes. It then uses the retained list to assemble responses from saved participant images, avoiding one coordinator payload query per participant. Canonical completion status and unresolved-participant checks remain. + +```text +Durable BEGIN -> prepare participants -> record receipts + COMMIT + | + resolve participants + record receipts + | + fetch saved participant images + | + durably acknowledge assembled images + | + release images + record cleanup receipts + | + SDK response +``` + +The signed two-Cell regression measures **five coordinator queries before this change and three afterward**. It checks response order, absent items, owner restoration, and subsequent writes. Oversized BEGIN payloads and transactions involving the legacy account Cell still use durable participant discovery. Token replay, uncertain replies, and recovery retain authoritative coordinator reads. Participant commands, durable decisions, saved-image reads, and cleanup receipts remain required. An earlier broader shortcut failed mixed-participant recovery verification and remains excluded; its investigation is retained. Removing queries is a measured reduction in work; it does not establish an end-to-end throughput gain or SQLite parity. + +### Reserve small replies for saved transaction images + +Each saved-image query first uses a 64 KiB compact reply envelope. It reads the immutable image captured by the participant, using the original coordinator identity and item position. A missing live item is a valid absent image; an unavailable saved image is an error for a committed response. + +```text +committed saved image -> compact query + | + +----------+-----------+ + | | + complete item WideRequired + | | + | existing wide query + | same saved identity + +----------+-----------+ + | + assemble response + | + durable acknowledgement + cleanup +``` + +A large image returns an explicit `WideRequired` marker, then uses the existing wide query. Errors do not trigger fallback. Items are never truncated, and the read does not substitute current live data. Wide fallback adds a query round trip. Reads remain serial within each participant so multiple large images do not reserve several wide replies together. + +This reduces reserved reply memory for small items. The HTTP peer receiver also retains its own request buffer, so total in-flight memory exceeds the 64 KiB reply alone. End-to-end throughput, large-item latency and SQLite parity require separate release measurements. Existing query opcodes and persisted transaction/image formats remain available; upgrades of roots created by older application releases still need separate qualification. + +## Latest release verification + +### Coalesced prepares: release qualification fails + +The prepare-batching release attempt measures source `2ebb948` with unchanged dependency pins and durability settings. **All 72 unique cases run**, including one declared unchanged single-Cell retry after the original trial fences during item seeding. The failed original trial remains in the report. The retry completes 6,117 requests with zero SDK errors; four Cells complete 170 requests with **77 errors across 17 cases**; SQLite completes 25,340 requests with zero errors. **SQLite parity and runtime qualification remain unmet.** + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 36.70 | 14.77 | 228.90 | +| PutItem | 11.14 | 0.00* | 261.38 | +| UpdateItem | 3.62 | 0.00* | 320.22 | +| TransactGetItems | 30.68 | 0.07* | 648.64 | +| TransactWriteItems | 0.84 | 0.00* | 203.80 | + +\* Cases contain errors; zero means no successful sample after owner/authorization failures, not healthy capacity. Single-Cell transaction writes complete eight calls in 9.49 seconds, p95 **8,378.14 ms**, versus SQLite's 1,022 calls and **118.60 ms**. Calls contain two items. Eight completions do not establish a tail distribution. Full tables retain all APIs, errors, counts and elapsed times. + +The 43-test verification proves 32 independent signed SDK transaction writes use **38–40 data commands instead of 64**, including recovery from a batch reply lost after publication. Results survive owner restoration. Formatting and strict Clippy pass. This is reduced publication work; the release attempt does not establish a throughput gain. + +Single-Cell SQL commands average **1.857 ms**, while follower durability responses average **356.278 ms**, peer lookup **193.687 ms** and enrollment **61.573 ms**. These scopes overlap and cannot be added. The original single trial and four-Cell run fence after directory-refresh waits of about **12.0/12.3 seconds**; the cause inside that phase remains unisolated. Reliable renewal and coordinator admission remain priorities. + +On 12 CPUs, SDK-window load is single retry **94.87→34.83**, four Cells **40.72→59.68**, SQLite **57.61→47.50**, with about **39.6–43.5 GiB of swap**. No task-local builds, tests or provider probes overlap measurement. Varying contention and ordering prevent causal attribution. Ten owned processes and three containers stop without forced BeyondDB cleanup; volumes remain retained. Full recovery, larger single-Cell budgets, live conversion and fleet capacity remain unqualified. + +### Scoped transaction-read metadata release + +The scoped-read release comparison measures committed source `eda759e` with unchanged reviewed dependencies. All **72 cases complete with zero SDK errors**: single Cell 12,116 completions, four Cells 18,859, SQLite 52,086. **SQLite parity remains unmet.** No API exceeds SQLite's eight-client rate in both BeyondDB modes. + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 83.75 | 308.51 | 785.56 | +| PutItem | 45.46 | 54.13 | 506.17 | +| UpdateItem | 62.02 | 41.04 | 495.30 | +| TransactGetItems | 105.41 | 0.98 | 598.33 | +| TransactWriteItems | 3.78 | 4.58 | 369.00 | + +Eight-client transaction writes complete 27/28/1,846 requests (single/four/SQLite), with successful p95 **2,338.10/2,472.83/81.39 ms**. Each call contains two items. The four-Cell read case completes only nine transactions in 9.14 seconds, with successful p95 **8,876.13 ms**. Low counts and short targets do not qualify sustained throughput or a tail distribution. Complete tables retain all APIs, errors, completions and elapsed times. + +The scoped optimization reduces fresh two-Cell read coordinator queries from five to three and passes 35 focused/library tests, formatting and strict Clippy. An earlier broader shortcut fails mixed account/data recovery and is excluded; its failures and baseline comparison remain in the verification record. This is a reduction in coordinator work, not an isolated end-to-end speedup. + +Four-Cell SQL command primitives average 0.800 ms, while follower durability command responses average 101.332 ms and durable follower append 24.537 ms. These scopes overlap, include different populations, and cannot be added. During concurrent transaction reads, peer lookup and enrollment average 87.868 and 49.800 ms. Discovery, durability and batching need further work. Runtime warnings record two short directory-refresh session-change failures; no terminal fencing occurs, and the earlier endpoint loss remains unresolved. + +On 12 CPUs, SDK-window host load is single **25.74→31.39**, four Cells **27.65→22.49**, SQLite **22.29→21.43**, with about 36–37 GiB of swap. No task-local build, tests or provider probes overlap measurement. Changing contention prevents causal attribution. All seven owned PIDs and two containers are absent; no BeyondDB process requires forced cleanup and volumes remain retained. The preceding head's [full recovery CI](https://github.com/crabbuild/beyonddb/actions/runs/37050870335) still fails missing GSI ownership and observed-stream EOF shutdown. Zero errors in this fixture do not close full recovery, old-root upgrades, larger single-Cell budgets, live conversion or SQLite parity. + +### Previous independent heartbeat release + +The independent heartbeat release attempt measures committed source `e1f7664` with unchanged reviewed dependencies. All **72 cases run**. Single-Cell BeyondDB and SQLite complete their 24 cases with zero SDK errors; four-Cell BeyondDB records **72 errors across 16 cases**, loses its endpoint and exits fenced. **SQLite parity and runtime qualification remain unmet.** + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 234.27 | 122.33 | 727.21 | +| PutItem | 31.42 | 0.00* | 751.03 | +| UpdateItem | 81.76 | 0.00* | 350.87 | +| TransactGetItems | 215.46 | 0.00* | 665.99 | +| TransactWriteItems | 4.54 | 0.00* | 278.76 | + +\* Cases contain errors after the four-Cell runtime fails; zero is not healthy capacity. Single-Cell transaction writes complete 29 requests with p95 **1,959.57 ms**, versus SQLite's 1,412 requests and **92.63 ms**. Calls contain two items. Full tables and raw results retain counts, errors, tails and actual elapsed time. + +The regression proves stalled log I/O held the heartbeat's shared mutex. The fix keeps a separate log-transition mutex and lets heartbeat renewal proceed; signed ETag retries preserve log state. Formatting, strict Clippy and 39 focused/library tests pass. The original fresh transaction diagnostic now completes 32 distinct writes with zero errors and verifies 64 items by strong reads. Resident coordinator waves are not consistently faster, so that diagnostic does not prove a coordinator-reuse speedup. + +**The fix does not resolve every fencing failure.** Four-Cell logs retain terminal Fenced errors; the precise remaining renewal phase is not yet isolated. SDK-window host load is single **33.50→28.05**, four Cells **52.76→69.93**, SQLite **69.45→54.44**, on 12 CPUs with about 36 GiB of swap. No local build, test or provider probe overlaps measurement. All ten fixture PIDs and three containers are absent; no forced cleanup is needed. Changing contention prevents attribution of an end-to-end speedup. Full recovery, larger single-Cell budgets, live conversion, sustained capacity and SQLite parity remain unfinished. The preceding source's [full SDK CI](https://github.com/crabbuild/beyonddb/actions/runs/37040618370) fails: native 48/48, peers 74/75 with missing GSI ownership, processes 7/8 with the observed-stream EOF shutdown panic. + +The follow-up renewal-phase diagnostic repeats the original four-Cell workload at source `7cba575`, adding operational failure observations with unchanged deadlines and durability. All 24 cases run: 8,502 completions, **six timeouts** (five transaction reads, one transaction write). No renewal warning or terminal Fenced error occurs; the earlier endpoint loss is not reproduced and remains unresolved. Eight-client transaction writes complete eight requests at 0.60 req/s with successful p95 7,613.65 ms. This is a diagnostic, with no fresh SQLite comparison or speedup claim. SDK-window load is 74.40→49.94 on 12 CPUs, with about 37 GiB of swap. Three PIDs and one container are absent without forced cleanup. A retained mailbox refusal and incomplete-resolution warning do not establish the cause of a specific timeout. + +### Previous Cellule member-expiry release verification + +The Cellule member-expiry release refresh measures committed source `61a74a4`, pinning reviewed Cellule `0f4ca09`. All **72 cases run**. Four-Cell BeyondDB and SQLite complete 24 cases each with zero request errors; single-Cell BeyondDB records **eight timeouts** in its concurrent transaction-write case. **The all-API SQLite performance goal remains unmet.** + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 55.65 | 63.56 | 509.12 | +| PutItem | 25.86 | 25.92 | 456.79 | +| UpdateItem | 30.93 | 41.83 | 400.27 | +| TransactGetItems | 82.34 | 2.83 | 526.00 | +| TransactWriteItems | 0.00* | 2.26 | 273.94 | + +\* No successful sample: all eight requests time out. Four-Cell transaction writes complete 17 requests with p95 **3,911.42 ms**, versus SQLite's **100.14 ms**. Calls contain two items. The complete tables retain every API, client count, tail, completion and error; raw JSON retains actual elapsed time. + +The upgrade includes private peer TCP_NODELAY, vectored writes, fresh-authority routing improvements, combined compaction/append publication, admitted owner fences and host member-expiry rotation. BeyondDB implements the required rotation callback using fresh enrollment and bounded signed fleet observations on the host's 30-second interval. Verification records passing focused authority, durability, prepare and single-Cell SDK cases, formatting, strict Clippy and 28 library tests. The broad two-owner test fails on a missing GSI owner. Preceding-source SDK CI passes native **48/48**, peers **74/75** and processes **7/8**, retaining coordinator ownership and EOF shutdown failures. No dependency source is patched; full recovery remains unqualified. + +**This sample does not isolate an end-to-end speedup or regression.** On 12 logical CPUs, SDK-window load is single **26.78→32.03**, four Cells **35.65→29.46**, SQLite **28.64→30.43**, with about **35–36 GiB of swap** in use. No task-local build, test or provider probe overlaps measurement. One mailbox-capacity warning occurs in single mode; four Cells retain two deferred transaction-resolution warnings. All seven fixture PIDs and both containers are absent without forced BeyondDB cleanup. Zero errors in four-Cell mode does not prove maintenance convergence or sustained capacity. Larger configurable single-Cell budgets, live conversion, old-root upgrades and SQLite parity remain unfinished. + +### Previous bounded prepare release verification + +The bounded prepare release refresh measures committed source `8b2f92e`. **Performance qualification fails:** the single/four-Cell fixtures lose their serving leases during measurement, record **26/71 request errors**, and complete no transaction writes. SQLite completes its 24 cases with zero errors after its missing release binary is rebuilt from the same pinned source. All **72 unique cases** run across the retained experiment and repaired SQLite invocation. **The all-API SQLite goal remains unmet.** + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 115.37 | 64.67 | 486.48 | +| PutItem | 12.45 | 0.00* | 297.95 | +| UpdateItem | 9.61 | 0.00* | 272.60 | +| TransactGetItems | 13.23 | 0.09* | 391.10 | +| TransactWriteItems | 0.00* | 0.00* | 239.73 | + +\* Cases contain errors. Zero means no request completed after endpoint loss, not healthy capacity. Four-Cell transaction reads complete one request and time out eight; its successful latency percentile cannot represent a latency distribution. The full tables retain all APIs, client counts, p95 values, completions and errors; raw JSON retains actual elapsed times. + +The held-publication regression proves **eight concurrent small prepares fit instead of three**. New account/data commands reserve 64 KiB replies for inputs up to 32 KiB; the prior roughly 4 MiB reply reservation exhausted the 16 MiB Cell mailbox. Complete failure images use the existing wide command only after a durable rejected receipt proves rollback. Reply loss retains authoritative recovery. Owner restoration, token reuse, mixed participants, capacity refusal, formatting, strict Clippy and 28 library cases pass. This is an admission improvement; the failed release run does not establish a transaction speedup. + +**This workstation sample does not isolate a source regression or speedup.** On 12 logical CPUs, SDK-window load rises 20.49→43.70 for single mode and 47.06→71.54 for four Cells; SQLite runs at 39.42→36.69. Swap is about 34.4–35.3 GiB for BeyondDB and 39 GiB for SQLite. No task-local build, test or provider probe overlaps measurement. The SQLite rebuild is separate and its new binary identity is recorded. All seven fixture PIDs and both containers are absent; volumes are retained. + +Owned BeyondDB logs end with `Fenced`; the immediate cause of delayed lease renewal is not isolated. Preceding-source full SDK CI also fails: native 48/48, peers 72/73, processes 7/8. Two-owner recovery, observed-stream EOF shutdown, full current-source recovery and old-root upgrades remain unqualified. Cellule stays at `a4500ad` for this measurement. Review of newer `origin/main` routing, TCP_NODELAY, compaction and member-expiry changes is preparatory; that upgrade is not qualified here. + +### Previous fresh remote routing release verification + +The fresh remote routing release comparison measures committed source `b285dae`. All **72 cases ran**, with one Single transaction-write throttling cancellation, one four-Cell transaction-read timeout and zero SQLite request errors. **The all-API SQLite performance goal remains unmet.** Only ListTables exceeds SQLite's eight-client request rate in both BeyondDB modes. + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 162.86 | 157.81 | 520.78 | +| PutItem | 37.71 | 18.14 | 137.04 | +| UpdateItem | 55.41 | 18.39 | 290.52 | +| TransactGetItems | 118.85 | 0.80 | 299.70 | +| TransactWriteItems | 0.72 | 1.97 | 403.02 | + +Transaction-write p95 is 9623.62/5522.40 ms for single/four Cells, versus SQLite's 46.15 ms. Completed requests are 7/16/2026, with one Single cancellation excluded from successful latency percentiles. Calls contain two items. The full report retains all API rates, tails, counts, errors and actual elapsed time. + +The gated regression proves fresh authority and owner enrollment reads overlap; neither read is skipped. A bounded session hint selects the probe, but fresh authority still decides ownership and lease expiry is rechecked after I/O. Changed owners get their own fresh enrollment read. The test checks current-owner probe failure, restoration on a third node and a signed strong read with SDK retries disabled. Existing drain/expiry cases, formatting, strict Clippy and all 28 library cases pass. + +**The measurement does not isolate an end-to-end speedup or regression.** Host load is single: 24.76→33.82; partitioned: 29.82→26.20; sqlite: 26.67→26.44, on 12 logical CPUs with about 42–43 GiB of swap in use. No task-local build, test or provider probe overlaps measurement. The temporary Python environment was recreated with the same boto3 version; the previous botocore version was not recorded. All seven fixture PIDs and both containers are absent, without forced BeyondDB cleanup. + +Single mode records a participant-capacity warning during SDK measurement and two follower-append warnings after it. Four-Cell owned logs are empty. The preceding source's [full SDK CI](https://github.com/crabbuild/beyonddb/actions/runs/36971574005) fails: native 48/48, peers 70/72 and process 7/8. Post-drain token replay, GSI ownership recovery and follower shutdown with the observed-stream EOF panic remain unresolved. Current-source full peer/process CI and old-root upgrades remain unqualified. + +Participant capacity evidence identifies the next hypothesis: prepare advertises a roughly 4 MiB maximum reply while the runtime reserves that bound against a 16 MiB per-Cell mailbox. A correctly bounded metadata-only prepare path needs a regression with real replies held in flight; failure images and durable transaction rules must remain. No prepare change is included in this release. + +### Previous follower discovery release verification + +The follower discovery release comparison measures committed source `d88f580`, with unchanged reviewed Cellule and ExtendDB pins. All **72 cases ran**, but four-Cell TransactGetItems has **nine SDK read timeouts**. Single mode and SQLite have zero request errors. **SQLite parity remains unmet; every eight-client API is below SQLite's request rate in this sample.** + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 91.90 | 17.28 | 565.38 | +| PutItem | 12.03 | 6.81 | 724.96 | +| UpdateItem | 20.22 | 8.44 | 730.31 | +| TransactGetItems | 19.04 | 0.07 | 437.19 | +| TransactWriteItems | 5.40 | 1.16 | 496.07 | + +Transaction-write p95 is 1692.89/7442.14 ms for single/four Cells, versus SQLite's 36.88 ms. Completed requests are 32/9/2485; requests contain two items. Four-Cell transaction reads complete zero/one requests at one/eight clients, with one/eight timeouts. Latency percentiles cover successful requests only and exclude failures. The full report retains rates, counts, errors, tails and actual elapsed time for every case. + +The transport regression proves two concurrent cold or expired follower lookups perform one signed fleet scan instead of two. Each follower still performs fresh canonical mTLS-bound authorization and fsyncs. The shared discovery survives only its in-flight callers; existing peer cache and hard advertisement expiry bounds remain. Cancellation, duplicate identities, durable recovery and store reopen are checked. Formatting, strict Clippy and 28 library tests pass. + +**This sample does not isolate an end-to-end speedup or regression.** The host is heavily contended: SDK-window load is 56.19→34.77 for single mode, 41.61→156.51 for four Cells and 149.14→62.17 for SQLite, on 12 logical CPUs with 34–36 GiB of swap in use. No task-local build, test or provider probe overlaps measurement. The failed experiment is retained; deadlines and retries are unchanged. All seven fixture PIDs and both containers are absent, with no forced BeyondDB cleanup. + +Both measurements retain two mailbox-capacity warnings; four-Cell shutdown adds two follower-append warnings outside the SDK window. No background convergence is claimed for timed-out requests. The preceding source's [full SDK CI](https://github.com/crabbuild/beyonddb/actions/runs/36826386923) is now terminal failure: native 48/48, peers 71/72, process 7/8. GSI ownership recovery and graceful shutdown with Cellule's observed-stream EOF panic remain unresolved. Current-source full peer/process CI and old-root upgrades remain unqualified. + +Read-locality evidence from the preceding release shows that a single base Cell often executes on a follower. Remote read routing and authorization are the next investigation; the GET counters include background work and do not identify each read's purpose. Single-Cell placement alone does not remove private peer I/O. + +### Previous cold catalog release verification + +The cold catalog release comparison measures committed source `1e8765c`, with unchanged reviewed Cellule and ExtendDB pins. All **72 cases have zero SDK request errors**. **SQLite write and transaction parity remains unmet.** Single mode exceeds SQLite's request rate only for eight-client ListTables; four Cells exceed it only for single-client DescribeTable. + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 257.50 | 475.01 | 1034.55 | +| PutItem | 49.97 | 42.92 | 731.88 | +| UpdateItem | 58.76 | 78.82 | 769.80 | +| TransactGetItems | 122.41 | 4.76 | 565.82 | +| TransactWriteItems | 6.14 | 3.06 | 515.66 | + +Transaction-write p95 is 1772.24/3469.94 ms for single/four Cells, versus SQLite's 36.97 ms. Completed requests are 35/19/2583; four-Cell single-client writes complete only three requests. Calls contain two items. The full report includes every API, client count, percentile and completed-request count. + +The signed regression proves cold coordinator catalog publication overlaps registration discovery and authority lookup. Bootstrap still requires the published catalog proof, and BEGIN still requires durable registration. Missing-generation checks, competing-owner fences, token identity and the resident fast path remain intact. Registration, incarnation restoration, token replay and the committed item are verified. + +**This sample does not establish an end-to-end speedup.** Transaction rates are below the preceding sample; ordinary writes and reads also vary. Host load is 28.64→30.53 for single mode, 30.45→52.43 for four Cells and 49.19→26.57 for SQLite, on 12 logical CPUs. Start snapshots report 31–33 GiB of swap in use. No task-local build, test or provider probe overlaps measurement. The old-release cold/warm diagnostic also shows substantial latency on warmed shards, but its sequential populations and changing host load do not isolate admission cost. + +Formatting, strict Clippy, 28 library tests and nine focused admission/registration cases pass. The native run passes 46/48; both failures pass individually without edits, and **the intermittent ownership/activity failures remain unresolved**. Preceding full SDK CI runs retain token replay, coordinator/GSI owner recovery and graceful shutdown failures, including the observed-stream EOF panic. Full peer/process CI and old-root upgrades remain unqualified. One incomplete-resolution WARN occurs during four-Cell measurement. All seven fixture PIDs and both containers are absent, without forced BeyondDB cleanup. Zero SDK errors does not establish maintenance convergence or sustained capacity. + +### Previous acknowledged-completion release verification + +The acknowledged-completion release comparison measures committed source `daa73a8`, with unchanged reviewed Cellule and ExtendDB pins. All **72 cases have zero SDK request errors**. **SQLite performance parity remains unmet; every eight-client API is slower than SQLite in this run.** + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 404.23 | 568.23 | 1323.04 | +| PutItem | 116.46 | 66.84 | 1210.74 | +| UpdateItem | 113.94 | 73.94 | 968.88 | +| TransactGetItems | 329.42 | 5.92 | 1177.13 | +| TransactWriteItems | 9.58 | 7.48 | 977.79 | + +Transaction-write p95 is 1074.01/1283.80 ms for single/four Cells, versus SQLite's 13.97 ms. Calls contain two items. The full report includes every API, client count, percentile and completed-request count. + +The signed regression proves fresh small transaction writes use **one coordinator query instead of four**. After an acknowledged fresh BEGIN and COMMIT, the adapter uses the exact accepted participant set and durably records every resolution receipt before success. Token replay, uncertain decisions, larger inputs and read-image cleanup retain authoritative readback. Lost participant replies require observed committed state; lost coordinator receipts return a transient error until replay confirms completion. Two-Cell values survive owner restoration. + +This reduces coordinator work but **does not demonstrate an end-to-end speedup**. The previous sample measured 10.76/10.15 transaction writes/s; this run measures 9.58/7.48. The ordinary read/write paths are unchanged and their rates also vary. Host load is 16.60→17.98 for single mode, 16.69→25.86 for four Cells and 25.63→14.56 for SQLite, on 12 logical CPUs. No task-local build, test or provider probe overlaps measurement. SQL handler means are 0.869/0.532 ms; durable follower append means are 14.657/17.394 ms. These overlapping background-inclusive populations are not additive request timings. + +Formatting, strict Clippy, 28 library tests, signed restoration and completion-fault cases pass. The full native run passes 47/48; the remaining new fixture's incorrect GetItem key is corrected and its focused rerun passes. All 48 pass across those two invocations. Full peer/process CI and older-root upgrades remain unqualified. The previous head's [full signed SDK CI](https://github.com/crabbuild/beyonddb/actions/runs/36815844420) subsequently fails: native 48/48, peers 70/71 and process 7/8. GSI owner recovery and killed-owner graceful shutdown fail; the latter again retains Cellule's observed-stream EOF panic. Three background WARN lines occur during measurement: two mailbox-capacity deferrals and one incomplete-resolution deferral. All seven fixture PIDs and both containers are absent, without forced BeyondDB cleanup. Zero SDK errors does not establish maintenance convergence or sustained capacity. + +### Previous returned-update release verification + +The returned-update release refresh measures committed source `a07acaf`, Cellule `a4500ad` and ExtendDB `7eaa89b`. It completes **72 cases with zero SDK request errors** across fresh single-Cell, four-Cell and SQLite fixtures. **SQLite write and transaction parity remains unmet.** + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 506.27 | 724.07 | 1630.70 | +| PutItem | 103.43 | 111.20 | 1287.67 | +| UpdateItem | 116.05 | 110.98 | 1067.54 | +| TransactGetItems | 412.00 | 7.39 | 1489.57 | +| TransactWriteItems | 10.76 | 10.15 | 868.46 | + +UpdateItem p95 is 97.97/112.52 ms for single/four Cells, versus SQLite's 17.56 ms. Transaction-write p95 is 977.59/919.62 ms, versus 18.65 ms. Transactions contain two items per call. The full report retains all APIs, client counts, completed requests and tail latency. + +**These are contended workstation samples.** The 12-core host's load falls from 39.26→25.84 during single mode, 25.35→22.68 for four Cells and 21.75→20.65 for SQLite. No task-local build, test or provider probe overlaps measurement. Read and transaction paths are unchanged, yet their rates rise too; the rate changes do not isolate a source speedup. SQL handler means are 0.876/0.609 ms, while follower durable append means are 16.368/20.103 ms. These overlapping populations include background work and must not be added into a request latency. + +The returned-update verification proves **64 concurrent signed updates use 24 durable commands instead of 64**. The pinned ExtendDB UpdateItem handler always requests the new image for capacity calculation, including `ReturnValues=NONE`; this previously bypassed the no-return batcher. Routed updates now coalesce distinct keys and return compact individual results. Conditions retain their own failures, successful writes retain individual stream ordinals, and repeated keys are deferred. The per-Cell queue has 64 permits, with batches capped at 16 updates, 1 MiB input and 128 KiB reply. A 2 ms partial-queue window trades a small single-client delay for sharing durable work. + +Large replies fall back to separate commands only after a confirmed rejected receipt proves that items, indexes and stream records rolled back. An ambiguous invocation is never retried internally. Signed tests verify same-key ADD results, condition-failure images, stream replay, owner restoration and 390,000-byte escaped item images. The same returned-image cases match the pinned SQLite server. Formatting, strict Clippy, 28 library tests, 48 native cases and three focused update tests pass. + +Cellule `a4500ad` contains website/documentation changes with identical Rust code to the previous pin. It does not fix runtime recovery or performance. The preceding source's [full SDK CI](https://github.com/crabbuild/beyonddb/actions/runs/36810977455) fails: native 48/48, peers 67/68, process 7/8. Coordinator ownership during restart and graceful shutdown after an owner kill remain open qualification gaps; the earlier post-drain replay failure does not recur in that run, without a claimed fix. The new source has not completed full peer/process qualification or old-root upgrade checks. Matching compiled peers are required. + +All seven fixture PIDs and both containers are absent. Neither BeyondDB fixture requires forced process cleanup; SQLite exits 0. Four-Cell logs retain two deferred transaction-resolution warnings during measurement. Zero SDK errors does not qualify maintenance convergence. A separate Cellule EOF guard proposal is unapplied and awaits the dependency approval required by AGENTS.md. + +### Previous acknowledged-BEGIN release comparison + +The acknowledged-BEGIN release refresh measures committed source `d46f6f9`, Cellule `e07670e` and ExtendDB `7eaa89b`. It completed **72 cases with zero SDK request errors** across fresh single-Cell, four-Cell and SQLite fixtures. SQLite write and transaction parity remains unmet. + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 125.36 | 324.92 | 1019.35 | +| PutItem | 70.63 | 66.97 | 639.97 | +| TransactGetItems | 85.82 | 5.43 | 913.05 | +| TransactWriteItems | 4.79 | 7.72 | 557.23 | + +Transaction-write p95 was 1990.78/1227.48 ms for single/four Cells, versus SQLite's 31.91 ms. Calls contain two items, so transaction item throughput is twice the request rate. + +**This is a contended workstation sample.** The 12-core host's load changed from 15.86→34.83 during the single run, 38.78→28.02 for four Cells and 26.97→22.00 for SQLite. No task-local build, test or provider probe overlapped measurement. Four-Cell transaction writes are higher than the preceding sample, while single-Cell writes are lower; this experiment does not isolate a source speedup. + +The signed SDK regression verifies fresh transaction coordinator queries fall from five to four. After a confirmed fresh BEGIN, the adapter reuses its exact accepted participants for inputs at most 32 KiB. Existing tokens, ambiguous replies, larger inputs and recovery still read durable coordinator state. All durable prepare, decision, resolution and completion checks remain. + +The full 48-case native suite, four signed Cell-model tests, formatting and strict all-target Clippy pass for this source. Its production-equivalent `bbe1803` [full SDK CI](https://github.com/crabbuild/beyonddb/actions/runs/36810977455) subsequently fails: native 48/48, peers 67/68 and process 7/8. Coordinator ownership during restart and graceful shutdown remain open. An earlier run also failed post-drain token replay and GSI ownership; no fix is claimed from their absence in this run. Older-root upgrades remain unqualified. All seven benchmark fixture PIDs and both containers are absent. + +### Previous coordinator-admission release comparison + +The coordinator-admission release refresh measures committed source `58ba0dd`, Cellule `e07670e` and ExtendDB `7eaa89b`. It completed **72 cases with zero SDK errors** across fresh single-Cell, four-Cell and SQLite fixtures, using signed AWS CLI creation and the unchanged boto3 harness. + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 237.59 | 679.86 | 786.23 | +| PutItem | 130.73 | 49.19 | 796.17 | +| TransactGetItems | 516.32 | 6.12 | 766.76 | +| TransactWriteItems | 9.35 | 5.65 | 539.37 | + +Single-Cell transaction reads measured p95 20.54 ms, versus 1816.47 ms with four Cells and 21.73 ms with SQLite. Single-Cell transaction writes measured p95 1126.21 ms, versus SQLite's 37.02 ms. SQLite parity remains unmet. + +The admission regression reduces fresh coordinator authority reads from two to one and four concurrent account registrations to one durable command. Metadata envelopes now reserve 4 KiB per input/result. Cancellation cannot acknowledge an unpublished registration, and acknowledged rows survive account owner restoration. Coordinator shard identities and token replay routing remain unchanged. + +This sample does **not** establish an end-to-end transaction-write speedup: the preceding record measured 10.39/7.55 requests/s, versus 9.35/5.65 here. Single-mode PutItem increased while four-Cell writes decreased. Fresh owner placement, I/O latency, host load and acknowledgment populations differ between fixtures. The count regression proves reduced admission work; the SDK samples do not isolate its throughput effect. + +Host load at SDK start/end was 15.37→13.67 for single, 13.34→14.53 for partitioned, and 14.89→23.91 for SQLite. No task-local build, test or provider probe overlapped measurement. All seven new server processes and both containers are absent. Four-Cell logs retain a deferred capacity sweep during measurement; single-mode follower-advertisement warnings occur during cleanup. Zero SDK errors does not qualify maintenance convergence or graceful shutdown. + +Focused checks, signed placement tests, 27 library tests, formatting and strict Clippy passed for this source. Subsequent [full SDK CI for production-equivalent `b9cbf50`](https://github.com/crabbuild/beyonddb/actions/runs/36806586859) failed: native 48/48, peers 65/67 and process 7/8. The latest qualification gaps are described above. Matching compiled peers are required, and older-root upgrades remain unqualified. + +### Previous placement release comparison + +The single/four-Cell release comparison measures source `8ecd6d5`, Cellule `e07670e`, and ExtendDB `7eaa89b`. Both BeyondDB fixtures use the same binary and four configured initial partitions, with `single` or `partitioned` selected at creation. AWS CLI creates each table; signed boto3 measures the unchanged workload. All 24 cases per fixture completed: **72 cases, zero SDK errors** across two BeyondDB modes and SQLite. + +| API, eight clients | Single req/s | Four Cells req/s | SQLite req/s | +| --- | ---: | ---: | ---: | +| GetItem | 345.76 | 525.60 | 946.39 | +| PutItem | 110.31 | 78.80 | 1106.55 | +| TransactGetItems | 336.73 | 6.20 | 724.66 | +| TransactWriteItems | 10.39 | 7.55 | 459.16 | + +Single-Cell transaction reads measured p95 33.14 ms, versus 1646.43 ms with four Cells. Single-Cell PutItem p95 was 97.21 ms, and transaction-write p95 was 920.8 ms; SQLite measured 15.27 and 47.44 ms. Tokenized writes still use the coordinator. The all-API SQLite objective remains unmet. + +Single mode returned 1,691 eight-client transaction reads, while the four-Cell sample returned 38; transaction writes completed 58/43. These short samples do not establish sustained capacity. One Cell also measured lower point-read throughput than four, so placement has workload tradeoffs. Host load, fresh owner placement, and provider latency differ between fixtures; the ratios do not isolate placement's service speedup. + +Host load was 17.93→16.75 for single, 17.46→23.30 for partitioned, and 24.15→26.65 for SQLite. No task-local build/test/provider probe overlapped measurement. Every new fixture PID/container is absent; volumes and scratch data are retained. Partitioned logs retain deferred recovery and mailbox-byte pressure in maintenance workers. Zero SDK errors does not qualify those workers' convergence. + +The placement verification passes three signed model tests, five creation/lifecycle tests, two statistics tests, 27 library tests, formatting and strict Clippy. The original record captured Rust CI success and SDK CI pending. Subsequent production-equivalent SDK CI failed as described above; full recovery and older-root upgrades remain unqualified. The new Single variant requires matching compiled peers. + +### Previous native-volume release pair + +The native-volume release pair +reuses source `7cf8f46` and the exact binary from the preceding host-bind run, +with Cellule `e07670e` and ExtendDB `7eaa89b`. RustFS stores data in a named +volume inside Colima. All 24 cases completed per backend with **zero SDK errors**. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 470.78 | 999.19 | 31.08 ms | +| PutItem | 61.89 | 760.40 | 221.77 ms | +| TransactGetItems | 5.22 | 771.38 | 2084.94 ms | +| TransactWriteItems | 8.01 | 544.07 | 1291.82 ms | + +Only DescribeTable exceeded SQLite at one/eight clients. Item operations and +transactions remain slower; the all-API performance objective is unmet. These +are five-second samples: eight-client transaction reads/writes completed 31/46 +requests, and do not establish a sustained capacity or production target. + +The prior host-bind pair measured PutItem at 22.27 requests/s and transaction +writes at 0.61 with five timeouts. This pair changed storage placement while +reusing the binary, but host load and fresh owner placement also changed. +BeyondDB host load was 25.33→27.14; SQLite ran afterward at 27.34→22.96. +No task-local build/test overlapped either measurement. The comparison cannot +isolate a code or storage speedup. All fixture processes/container are absent; +the named volume and server scratch data remain available for inspection. + +A direct S3 diagnostic +completed 48 cases without errors, using opposite storage orders. Native-volume +1-KiB conditional replacements reached 226–238 requests/s at eight clients, +versus 71–77 on binds. GET throughput was lower in those native samples while +load varied. Keep provider storage placement explicit in benchmark results. +The [deployment example](deployment.md#local-object-store-fixture) already uses +a named volume. + +Runtime means were SQL command 0.542 ms, provider GET/PUT 6.492/27.109 ms, +and publication total 146.625 ms. Across 3,610 samples per follower phase, +lookup averaged 6.140 ms, round trip 30.778 ms, fresh enrollment 12.447 ms, +and durable append 17.620 ms. These scopes overlap and include background work; +they must not be summed into SDK latency or treated as a causal decomposition. +The full record retains every API, counts, phases, source/binary hashes and host +snapshots. Full signed SDK recovery CI failed on this source: native 48/48 +passed, peers 57/59 and process 7/8 failed. Full recovery remains unqualified. + +### Previous bounded-snapshot host-bind release pair + +The bounded coordinator snapshot release pair +measures source `7cf8f46`, Cellule `e07670e`, and ExtendDB `7eaa89b`. +All 24 cases completed per backend. BeyondDB recorded **five SDK timeouts**, +all in eight-client TransactWriteItems; SQLite recorded zero. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 323.31 | 745.57 | 62.65 ms | +| PutItem | 22.27 | 723.85 | 711.47 ms | +| TransactGetItems | 0.98 | 830.83 | 8614.57 ms | +| TransactWriteItems | 0.61 | 990.81 | 6829.62 ms* | + +\* Nine transaction writes completed; five timeouts are excluded from the +percentile. Only single-client ListTables exceeded SQLite. Every eight-client +case remained slower; the all-API SQLite objective is unmet. + +Host load was 37.42→32.62 during BeyondDB, then 32.57→26.22 during SQLite on +12 logical CPUs. No task-local build/test overlapped measurement. Fresh owner +placement and provider costs also vary. This sample does not isolate the query +fusion's throughput effect or establish production/fleet capacity. All fixture +PIDs and the exact RustFS container are absent, with no forced PID cleanup. + +The counted driver regression +fails before the change with seven coordinator queries and passes after with +four. One bounded query observes durable status and small immutable payloads; +large inputs retain chunked retrieval. Durable preparation, decision, resolution +and final-status checks remain. The new query changes the compiled coordinator +contract; older-root upgrades remain unqualified. + +Local gates pass: 12 transaction cases, restart read cleanup, seven signed +coordinator SDK cases, 27 library tests, formatting and strict all-target Clippy. +The focused count regression overlaps the transaction selection. Source Rust CI +passed; full SDK CI subsequently failed: native 48/48 passed, peers 57/59 and +process 7/8 failed. The additional large remote read root assertion remains +unqualified. Prior-source SDK CI on the +same Cellule pin failed the global-index owner and graceful-stop assertions: +native 47/47 passed, peers 58/59 and process 7/8 failed. Full recovery is open. + +SQL command execution averaged 0.384 ms, provider GET/PUT 14.825/261.135 ms and +publication total 1271.151 ms. Follower lookup averaged 82.857 ms across 1,488 +samples; HTTP round trip 49.520 ms, enrollment 33.808 ms and durable append +15.005 ms each had 1,486 samples. These overlapping populations include +background work and in-flight operations; counts need not match, and event +times must not be added into SDK latency. The full record retains every API, +sample count, failure, phase metric, source/binary hash and host snapshot. + +### Previous Cellule e07670e release pair + +The Cellule e07670e release pair +measures source `664e07a`, reviewed upstream Cellule `e07670e`, and ExtendDB +`7eaa89b`. All 24 cases completed per backend. BeyondDB recorded **seven SDK +timeouts** in eight-client TransactGetItems; SQLite recorded zero. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 310.54 | 459.75 | 64.82 ms | +| PutItem | 10.09 | 800.97 | 1880.62 ms | +| TransactGetItems | 0.19 | 752.99 | 1429.53 ms* | +| TransactWriteItems | 1.14 | 453.42 | 8179.69 ms | + +\* Only two transaction reads completed; seven timeouts are excluded from the +percentile. This p95 does not establish improvement over an error-free run. +Only single-client Query exceeded SQLite. Every eight-client case remains slower; +the all-API SQLite objective is unmet. + +The workstation was heavily contended: BeyondDB host load was 69.98→23.59 on +12 logical CPUs; SQLite ran afterward at 22.59→21.72. No local build or test +from this task overlapped measurement. Fresh owner placement and provider costs +also vary. This pair does not isolate the upgrade's performance effect or qualify +production/fleet capacity. All fixture processes and the exact RustFS container +are absent, with no forced PID cleanup. Raw data and hashes are retained. + +The new main includes the approved one-read append API unchanged, lease-fenced +resident-route reuse, bounded owner-discovery coalescing, read-replica release +checks and serialized directory-cache index snapshots. The lockfile changes only +seven Cellule Git sources. Upgrade verification +passes 27 library tests, 58 residency/routing tests, formatting and strict Clippy; +seven coordinator cases also pass and overlap residency. Upstream's four CI +workflows and BeyondDB source Rust CI pass. The subsequent full signed recovery CI failed: native 47/47 passed, peers +58/59 and process 7/8 failed. Recovery qualification remains open. + +Across 1,572 samples per follower phase, lookup averaged 85.002 ms, HTTP round +trip 45.940 ms, fresh enrollment 29.041 ms and durable append 16.119 ms. SQL +command execution averaged 0.423 ms, provider GET/PUT 15.753/264.510 ms and +publication total 1240.378 ms. These overlapping populations include background +work; do not add them into SDK latency. The complete report includes all APIs, +request failures, case metrics and fixture metadata. + +### Previous parallel coordinator release pair + +The parallel coordinator release pair +measures source `388ae62`, reviewed Cellule `8ca658b`, and ExtendDB `7eaa89b`. +Both backends completed all 24 cases with **zero SDK errors**. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 357.58 | 870.42 | 36.02 ms | +| PutItem | 22.97 | 912.31 | 882.76 ms | +| TransactGetItems | 1.06 | 893.41 | 7560.11 ms | +| TransactWriteItems | 1.25 | 673.41 | 6953.71 ms | + +Only single-client DescribeTable and ListTables exceeded SQLite. Every +eight-client case remained slower. Host load was 21.34→18.98 during BeyondDB +and 18.82→19.52 during SQLite. Owner placement and provider latency also vary. +Point reads slowed despite their unchanged path, so this sequential pair does +not isolate the admission change's effect. The all-API SQLite objective remains unmet. + +The concurrent signed SDK regression +proves that two independent cold coordinators can enter authority creation +concurrently. It failed before the change and passes afterward, with durable +registration and data-owner restoration checks retained. Bounded pending-slot +accounting protects capacity; cancellation releases its reservation. Published +roots and reclamation continue through exclusive admission. + +Seven coordinator cases and five reclamation cases pass with CI fixture +scheduling, as do all 27 library tests, formatting and strict Clippy. A parallel +fixture run failed replay after drain; its cause remains open. Earlier full SDK +CI on `a18652a` passed native 47/47 but failed peers 55/56 and process 7/8. +Full signed recovery qualification remains open. + +Across 1,918 samples per phase, peer lookup averaged 32.302 ms, HTTP round trip +36.595 ms, fresh enrollment 21.180 ms, and durable append 14.768 ms. Provider +GET/PUT means were 10.430/221.287 ms and publication total was 1099.160 ms, +compared with SQL command execution of 0.430 ms. These overlapping populations +include background work; they cannot be added into SDK latency. The complete +record retains all cases, sample counts, host metrics, source and binary hashes, +and verified cleanup. + +### Previous follower-phase release pair + +The follower-phase release pair +measures source `22002dc`, reviewed Cellule `8ca658b`, and ExtendDB `7eaa89b`. +Both backends completed all 24 cases with **zero SDK errors**. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 616.61 | 1307.63 | 20.03 ms | +| PutItem | 61.24 | 1565.61 | 280.96 ms | +| TransactGetItems | 2.46 | 1128.43 | 4472.35 ms | +| TransactWriteItems | 1.83 | 1131.71 | 4676.52 ms | + +Only single-client ListTables exceeded SQLite. BeyondDB host load fell from +16.94 to 13.58; SQLite ran afterward at +13.58→11.78. Fresh owner placement also differs between fixtures. +This instrumentation revision does not establish a speedup or fleet capacity. + +The new append observations recorded 2,948 samples per phase across all nodes: +peer lookup mean 8.463 ms, HTTP round trip 26.766 ms, fresh enrollment +12.071 ms, and durable append 14.247 ms. Receiver phases overlap the round trip; +these event populations include background work and cannot be summed into SDK +latency. SQL command execution mean was 0.386 ms, provider PUT mean 86.542 ms, +and publication total mean 416.128 ms. The evidence supports examining cold +transaction admission and object publication next, while retaining exact fences +and fresh authorization. It does not isolate a network-only duration. + +Phase verification +passed all 27 library tests, formatting and strict Clippy, including actual mTLS +append/reopen phase counts and cancellation accounting. Full signed peer/restart +qualification and the all-API SQLite objective remain open. + +### Previous resident-admission release pair + +The resident-admission release pair +measures source `6e92b5d`, Cellule `8ca658b`, and ExtendDB `7eaa89b`. All 24 +cases completed per backend: BeyondDB recorded **four SDK errors** in +eight-client TransactGetItems; SQLite recorded zero. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 307.70 | 818.16 | 86.21 ms | +| PutItem | 21.17 | 1102.79 | 877.43 ms | +| TransactGetItems | 0.48 | 759.27 | 7441.80 ms* | +| TransactWriteItems | 1.74 | 783.82 | 5192.66 ms | + +\* Successful calls only; four failed requests are excluded. SQLite was faster +in every measured case. Host load during BeyondDB was 41.43→32.36; SQLite ran +sequentially at 32.36→20.40. Fresh owner placement also varies. This busy local +sample cannot isolate the optimization's speedup or establish fleet capacity. + +The focused admission regression +proves four warm admissions remove four canonical coordinator reads and perform +zero provider reads. Drain, a new provisioner, and missing registration still +use canonical admission. Two focused regressions, five coordinator lifecycle +checks, formatting, strict Clippy, and source Rust CI pass. Full signed peer/restart +qualification remains open. Provider and follower costs remain the next profiling +focus: SQL command/query means were 0.456/0.115 ms, logical provider PUT mean +181.040 ms, and fleet command-response mean 149.089 ms. These overlapping +populations must not be summed into SDK latency. The all-API SQLite objective +remains unmet. + +### Previous provider-observation release pair + +The provider-observation release pair +measures source `b2b6350`, Cellule `8ca658b`, and ExtendDB `7eaa89b`. Both +backends completed all 24 cases with **zero SDK errors**. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 655.57 | 1305.32 | 18.73 ms | +| PutItem | 57.05 | 1178.69 | 325.14 ms | +| TransactGetItems | 2.19 | 1213.74 | 5557.30 ms | +| TransactWriteItems | 2.13 | 1019.84 | 3922.09 ms | + +Only single-client DescribeTable exceeded SQLite. Every eight-client case +remained slower. Host load fell from 16.96 to 13.70 during BeyondDB and from +13.70 to 11.66 during SQLite; fresh owner placement also varies. This sample +does not establish a speedup from the discovery recovery fix or fleet capacity. + +The new provider metrics report GET mean 3.972 ms and PUT mean 84.768 ms across +the snapshot interval, while SQL primitive command/query means were +0.358/0.103 ms. These are overlapping events with different counts, including +background work; do not sum them into SDK latency. The full report contains +all 24 cases, raw results, metrics, host metadata, hashes, and verified cleanup. +The focused discovery regression +passes, but the full signed peer/restart scenario still fails. Production +recovery qualification and the all-API SQLite objective remain open. + +### Previous one-read follower append release pair + +The one-read follower append release pair +measures source `9b62001` and reviewed Cellule `8ca658b` +([dependency PR](https://github.com/crabbuild/cellule/pull/31)). ExtendDB remains +`7eaa89b`. All 24 cases completed per backend; BeyondDB recorded **four SDK +errors** in eight-client TransactGetItems, while SQLite recorded zero. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 302.92 | 889.90 | 61.61 ms | +| PutItem | 27.24 | 541.20 | 788.70 ms | +| TransactGetItems | 0.62 | 685.53 | 7,935.19 ms* | +| TransactWriteItems | 1.18 | 691.08 | 7,293.37 ms | + +\* Successful calls only; four failed requests are excluded. The report contains +all APIs, both client counts, completed/error counts, runtime metrics, host +snapshots, artifact hashes, and cleanup evidence. Only single-client +DescribeTable and ListTables exceed SQLite in this sample. All eight-client +rates remain below SQLite. The all-API performance objective is unmet. + +The approved change removes one consecutive canonical enrollment read per +follower append while retaining identity, scope, lease, log authority, and fsync +checks. Its counted-store and signed SDK process-kill tests pass. Host load was +19.02→19.03 during BeyondDB and 19.03→20.34 during SQLite; fresh owner placement +also varies between fixtures. This local sample does not establish a service +speedup. Cellule's four CI workflows and BeyondDB Rust CI passed. Full SDK CI +failed one long peer recovery test; native SDK and process suites passed. The +recovery follow-up +records the focused fix and the remaining full-scenario failure. + +### Previous resident-routing release pair + +The resident-routing release pair +measures source `c6fb584`, still pinned to Cellule `70bd25f` and ExtendDB `7eaa89b`. +All 24 cases completed per backend. BeyondDB recorded two SDK errors in the +eight-client TransactGetItems case; SQLite recorded none. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 359.56 | 975.69 | 58.69 ms | +| PutItem | 42.26 | 523.24 | 421.49 ms | +| TransactGetItems | 0.78 | 595.03 | 9,243.71 ms* | +| TransactWriteItems | 1.47 | 528.34 | 5,746.33 ms | + +\* Successful requests only; two failed requests are excluded. See the report +for all APIs, both client counts, sample sizes, first-error details and raw logs. +DescribeTable and ListTables exceed this SQLite sample at both client counts; +item operations and transactions remain slower. The all-API objective is unmet. + +Host load was 17.02→21.29 during BeyondDB and 21.19→18.06 during SQLite, with +roughly 19.6 GiB of allocated swap on the 12-CPU workstation. These sequential +samples do not isolate a code improvement from host conditions. Native SDK and +process CI passed, but one long peer recovery test failed; full qualification +remains open. The approved one-read follower append change is not included in +this measured release. + +### Previous Cellule upgrade release pair + +The Cellule `70bd25f` release rerun +uses source `37b25ff`, upgraded from Cellule `30671d5` to the latest `origin/main` +checked on September 30. All six direct dependencies and seven lockfile packages +pin the new revision; ExtendDB remains `7eaa89b`. The measured code also adds the +500 ms backend table-key metadata cache for batch APIs, so this is a combined +product/dependency sample. + +Both backends completed all 24 cases. BeyondDB recorded **25 SDK errors**; +SQLite recorded zero. Each failing case’s retained first error was a read timeout. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 100.58 | 415.51 | 389.14 ms | +| PutItem | 12.28 | 261.16 | 1,844.92 ms | +| TransactGetItems | 0.09 | 297.73 | 651.35 ms* | +| TransactWriteItems | 0.00 | 221.13 | — | + +\* The transaction-read percentile is a single successful call, excluding eight +timeouts. All eight transaction writes timed out. Eight batch writes at eight +clients and one transaction write at one client also timed out. The report has +all API rates, sample counts, latencies, raw results, runtime snapshots, and fixture +settings. + +Host load rose from 28.35 to 52.46 during BeyondDB, then fell from 52.86 to 38.85 +during SQLite, on 12 logical CPUs with about 19.4–19.5 GiB of swap in use. This +checkout had no build or test running during measurement; other work continued. +These sequential runs do not establish a controlled speed improvement, a regression, +or production capacity. The all-API SQLite objective remains unmet. + +The locked release build, 25 library tests, backend-cache regression, formatting, +strict Clippy, and Rust CI passed. Full SDK CI on the measured source failed: +native 45/47, peers 51/52, process 8/8. Two native fixtures warmed the new route +cache before blocking control reads; both failures reproduced locally and passed +with their original assertions after giving recovery a fresh client. This test-only +correction happened after measurement. The follow-up CI run passed native 47/47 +and process 8/8 but failed two peer cases: directory retirement saw a draining +Cell, and abandoned-coordinator recovery missed its deadline. A focused credential +pressure regression now passes after one bounded retry of a proven not-started +query. The original long test passed that step, then failed coordinator recovery. +The opt-in resident resolver also avoids authority reads after handle-cache expiry +and checks the current actor before reusing a cached handle. Signed owner-expiry +recovery passes with caches off and on. These source changes are now measured in the resident-routing pair above; +see the follow-up verification record. +Two local process-suite attempts failed owner recovery and subsequent fixture +creation; Docker’s filesystem had almost no free inodes. All eight process tests +passed in CI. Full recovery qualification remains open. + +### Previous instrumented release pair + +The instrumented release rerun +uses source `8e33705` and Cellule `30671d5`, which was current at measurement. Both backends +completed all 24 cases: BeyondDB recorded **38 SDK errors**, SQLite zero. Each +failing case's recorded first error was a read timeout. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 48.82 | 896.78 | 555.58 ms | +| PutItem | 0.47 | 104.16 | 6,004.70 ms* | +| TransactGetItems | 0.27 | 401.82 | 8,635.70 ms* | +| TransactWriteItems | 0.10 | 222.09 | 8,877.42 ms* | + +\* Percentiles exclude errors. Only seven puts, three transaction reads, and +one transaction write succeeded in these eight-client cases. + +The 12-CPU, 32 GiB workstation was heavily contended: one-minute load was +84.28–84.45 during BeyondDB and 75.27–49.92 during SQLite, with about 20–21 GiB +of swap in use. Local builds and tests finished before measurement; other work +continued. These sequential samples do not establish a controlled speed ratio, +a code regression, or production capacity. The all-API SQLite objective remains unmet. + +Runtime counters recorded 288 follower-backed command replies and two object-backed +replies. Mean command worker time was 6.34 ms, queue time 188.94 ms, and object +publication 2,790.98 ms. These observations include seeding/background work and +different overlapping events; they cannot be summed into SDK latency. Product +routing, peer metadata reads, and bootstrap provisioning are not covered by +these runtime counters in this fixture. The report retains case-boundary snapshots, +all API rates, sample counts, errors, and host memory observations. + +Fresh local checks passed all 46 native integration tests, 25 library tests, +both signed durability controls, actual owner process-kill recovery, formatting, +strict Clippy, and the locked release build. Rust CI passed; full SDK CI on the +measured source failed with native 46/46, peers 51/52, and process tests 8/8. +Abandoned-coordinator recovery after the remote owner stopped renewing missed +its 45-second deadline. Recovery qualification remains open. The follower recovery fixture now waits +for its warmup receipt to publish and requires activation on the first write +whose object publication is blocked. + +### Previous follower diagnostic pair + +At that measurement, Cellule `origin/main` was `30671d5`, used +by all direct dependencies and lockfile packages. The fresh release rerun +uses measured source `9cba5f1` (production code `d04176c`) and completed all +24 cases per backend. BeyondDB had **three transaction read timeouts**: +one in `TransactGetItems`, two in `TransactWriteItems` at eight clients. +SQLite had zero errors. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 186.05 | 449.22 | 103.82 ms | +| PutItem | 11.96 | 415.28 | 2,253.55 ms | +| TransactGetItems | 0.80 | 470.92 | 9,089.13 ms* | +| TransactWriteItems | 0.62 | 321.38 | 9,072.64 ms* | + +\* Percentiles exclude timeouts; only nine transaction reads and eight writes +succeeded in those cases. Host load rose from 14.31 to 25.59 during BeyondDB +and from 28.81 to 29.49 during SQLite on 12 logical CPUs. No local build or +test overlapped the measurements. Placement, IAM, and durability contracts +also differ, so this pair does not establish a controlled speed ratio or +production capacity. The all-API SQLite objective remains unmet. + +The release build, 25 library tests, 11 transaction tests, formatting, and +strict Clippy passed. Full SDK CI on the production code failed: elastic tests +46/46, peers 51/52, and process tests 7/8. Credential lookup failed after the +explicit memory-exhaustion probe, and the follower test did not activate its +log during unblocked warmup. Recovery qualification remains open. Four background +operations were deferred by Cell mailbox-byte capacity during measurement. +The follower closure was not reproduced; append errors occurred only after +measurement during cleanup. The report retains all API rates, latencies, +errors, fixture metadata, and raw logs. + +### Previous combined coordinator release pair + +The combined coordinator-commit release pair +uses source `3ccab15`, Cellule `30671d5`, three BeyondDB processes, four initial +partitions, experimental follower durability, and signed boto3. Both backends +completed all 24 cases. BeyondDB recorded **eight timeouts**: five in eight-client +`TransactGetItems`, three in eight-client `TransactWriteItems`. SQLite had zero errors. + +| API, eight clients | BeyondDB requests/s | SQLite requests/s | BeyondDB p95 | +| --- | ---: | ---: | ---: | +| GetItem | 234.19 | 778.00 | 66.49 ms | +| PutItem | 23.57 | 477.86 | 645.39 ms | +| TransactGetItems | 0.50 | 525.13 | 7,889.40 ms* | +| TransactWriteItems | 0.56 | 458.64 | 8,429.05 ms* | + +\* Transaction percentiles exclude timeouts. Only six reads and eight writes +succeeded in the eight-client cases. One-minute host load changed from +27.75 to 44.72 during BeyondDB and from 45.67 to 51.88 during SQLite, on +12 logical CPUs. These sequential samples do not establish a controlled +speed improvement or production capacity. The all-API SQLite objective remains unmet. + +The successful transaction path now records participant prepare receipts and +COMMIT in one coordinator command. Its regression verifies exactly one durable +coordinator commit, idempotent replay, rejected incomplete/wrong receipts, and +restoration from the published root before participant resolution. All 25 +library tests, 11 transaction tests, formatting, strict Clippy, the release +build, and Rust CI passed. This proves the command reduction; it does not +prove an end-to-end speedup. + +Full SDK CI on `3ccab15` **failed**: elastic Cells 46/46, peers 51/52, process +tests 7/8. A two-owner recovery test failed its index-owner assertion and the +follower durability test never activated a log. Broad SDK owner restart passed +in CI; a local retry committed transactions but missed the replacement server's +45-second health deadline. The measured frontend also logged 12 node-log +submission rejections with `RuntimeClosed` and fallback to object coverage. +The dependency pin and follower durability remain incompletely qualified. +A diagnostic follow-up +adds first-error logging and records two passing local follower process-kill +checks. The CI activation failure remains unresolved; these checks do not +change the benchmark results. + +The preceding compact transaction-read pair +recorded six transaction-read timeouts. That response codec keeps legal large +binary and escaped-string reads on the single-Cell query path, avoiding JSON +expansion into durable saved images. Signed remote-owner reads preserved exact +values without mutating the participant root; large binary reads also survived +an actual owner restart. Query codec versions are now 3; mixed-version peer +rollout is unqualified. Its full SDK CI failed with elastic Cells 46/46, +peers 50/52, and process tests 7/8. See both reports for raw measurements, +sample counts, CI links, and local evidence. + +The preceding peer owner-cache pair +recorded 19 timeouts under different load. Its signed mTLS regression proves +that the opt-in 500 ms private receiver cache removes repeated resident +handle authority reads, refreshes after expiry, and rejects unauthorized or +drained-owner requests. That focused proof does not establish end-to-end +SQLite parity. + +## Earlier object-publication fixture + +The release repeat after the peer read fix +uses BeyondDB `832da2a`, Cellule `30671d5`, and pinned ExtendDB SQLite. Both +backends completed all 24 signed API/client cases with **zero foreground +errors**. At eight clients, BeyondDB/SQLite measured 857/778 `GetItem`, +16.9/746 `PutItem`, 1.33/576 `TransactGetItems`, and 1.20/371 +`TransactWriteItems` requests/s. BeyondDB logged three deferred background +sweeps from mailbox capacity. RustFS used a fresh bind mount in Colima's +shared home directory after the Docker VM ran out of space; the preceding +named-volume attempt +became fenced and recorded 45 foreground errors. Host load and storage paths +differ between runs, so these rates do not prove a code-driven improvement. + +The peer read fix passed a signed remote-owner regression and the large +binary/escaped transaction checks after owner restart. The longer recovery +test subsequently failed an index-owner assertion with `owner=None` after +the signed index query and journal acknowledgement converged. That full SDK CI run did not pass. Zero foreground errors in this benchmark +does not mean recovery qualification or the all-API performance target is met. + +The initial signed-API rerun on this pin used Cellule `30671d5` and ExtendDB SQLite +to `7eaa89b`. Both harnesses completed all 24 five-second cases. At eight +clients, BeyondDB/SQLite measured 327/721 `GetItem`, 13/481 `PutItem`, and +2.11/277 `TransactWriteItems` successful requests/s. BeyondDB's eight-client +`TransactGetItems` case also had **seven read timeouts**; the other 23 cases +had zero foreground errors. The server logged five deferred background +operations from Cell mailbox-byte exhaustion. One-minute load on the +12-logical-CPU host changed from 30.2 to 48.1 during BeyondDB and ended at +27.0 after SQLite. See the full table, p95 latencies, fixture, and raw +JSON. These sequential +samples do not establish a controlled speed ratio or production capacity. + +The [preceding complete Cellule `9e17746` rerun](../benchmarks/2026-09-29-cellule-main-rerun/README.md) +finished all 24 cases for each backend with zero foreground errors; it is +historical context rather than a matched baseline for the new pin. An +[earlier attempt at that pin](../benchmarks/2026-09-29-cellule-main-attempt/README.md) stopped after ten BeyondDB cases when eight-client TransactGetItems returned -a throttling cancellation. SQLite completed all 24 cases while host load rose -to 90.4 on 12 logical CPUs. A signed 1 KiB PutItem survived an unclean owner -restart, while the larger SDK restart suite separately failed with HTTP 503 -during GSI setup under similar contention. That verification failure remains -open pending a successful repeat. +a throttling cancellation. On the `9e17746` pin, GitHub Actions later passed +all seven server-binary restart tests and 47 of 48 peer-network tests. On the +new `30671d5` pin, the first GitHub qualification run passed all 46 elastic +tests, 46 of 48 peer-network tests, and six of seven server-binary tests. +Concurrent-delete placement, a two-owner large binary transaction read, and +an oversized same-Cell read failed. Later qualification runs are recorded above; the +new pin remains incompletely qualified. ## Refresh of the earlier high-throughput sample @@ -68,9 +1014,9 @@ signed AWS SDK request With the default `auth_cache_enabled: false`, the server has no process-wide credential or authorization-result cache. It keeps a positive in-memory proof that a credential Cell exists, while it still reads the credential record for every signed request. Table metadata and authorization use pass-through stores so concurrent deletion, recreation, and policy changes are observed. Item routing reads the published directory; writes also wait for durable publication. These choices protect correctness but add work compared with an embedded SQLite test server. The route anchor and leaf can be read concurrently because the anchor remains the authority for whether a route is published. -For a workload that accepts bounded cross-node visibility, set `auth_cache_enabled` to `true` in the node configuration. This enables ExtendDB's 60-second stale-while-revalidate credential, IAM, and table metadata caches and wires local management invalidation through `AuthCacheRegistry`. It also caches immutable catalog proofs and resident local Cell handles for 500 ms, so warm requests avoid repeated catalog and owner-resolution reads. The Cell handle still fences drained owners; authority is refreshed after the cache window. Keep it disabled when immediate remote credential revocation, table recreation visibility, or owner changes are required. +For a workload that accepts bounded cross-node visibility, set `auth_cache_enabled` to `true` in the node configuration. This enables ExtendDB's 60-second stale-while-revalidate credential, IAM, and table metadata caches and wires local management invalidation through `AuthCacheRegistry`. It also caches immutable catalog proofs and resident local Cell handles for 500 ms. After expiry, the runtime can resolve the resident actor without provider reads. Cached handles are checked against the current resident actor before reuse. Handles still fence drained owners; remote routing, admission and recovery use exact authority checks. Keep the flag disabled when immediate remote credential revocation or table recreation visibility is required. -The same opt-in mode caches positive `DescribeTable` and `ListTables` responses for 500 ms, with at most 128 entries of each type per node. Local table creation, deletion, and update invalidate these responses as soon as the durable command completes. A remote node's table change can remain absent from a cached response until its entry expires. Item reads and writes still reach their owning Cell. +The same opt-in mode caches positive `DescribeTable`, `ListTables`, and backend table-key metadata for 500 ms, with at most 128 entries of each type per node. Batch APIs call the backend metadata lookup directly, so this cache also removes repeated account-Cell queries on that path. Local table creation, deletion, and update invalidate these entries as soon as the durable command completes. A remote node's table change can remain absent from a cached response until its entry expires. Item reads and writes still reach their owning Cell. On a fresh release fixture, the metadata cache measured 489/571 `DescribeTable` and 583/786 `ListTables` requests/s at one/eight clients, compared with 285/384 and 215/472 in an earlier BeyondDB fixture. A nearby SQLite fixture reached 835/1,196 and 701/343 respectively; its eight-client listing rate fell under host contention. The [raw metadata sample](../benchmarks/2026-09-29-metadata-cache/README.md) records the conditions. These short runs show an improvement in the cached path, not consistent SQLite parity. @@ -97,7 +1043,7 @@ The server now sizes Cellule's SQL worker pool from host parallelism (capped by When `auth_cache_enabled` is enabled, routed table directory leaf pages are cached by account and table generation. Large directories can retain up to 64 partial pages instead of caching only a complete page. The owning data Cell still checks the cached epoch, and stale or split routes invalidate the entry; this keeps route changes safe while avoiding a directory traversal on steady-state point operations. The local handle cache is 500 ms. An earlier release fixture measured `GetItem` at 1,323.6 requests/s with one client (p95 1.4 ms) and 1,776.6 requests/s with eight clients (p95 7.4 ms), `Query` at 1,293.9/1,722.2 requests/s (p95 1.4/7.6 ms), `PutItem` at 62.6/120.8 requests/s (p95 25.4/155.2 ms), and `UpdateItem` at 54.4/103.6 requests/s (p95 68.0/170.1 ms). All cases had zero errors. In that earlier run, reads exceeded the one-client SQLite samples; durable writes remained slower because each mutation waited for publication. -Increasing the local handle cache from 50 ms to 500 ms reduces repeated authority resolution inside a transaction. On a fresh four-partition fixture, signed `TransactWriteItems` reached 4.46 requests/s at one client (p95 306 ms) and 1.55 requests/s at four clients (p95 2.87 s), with zero errors. An eight-client run reached 2.02 requests/s before one `ServiceUnavailable`; the coordinator and durable participant Cells still contend under concurrency. The longer cache remains safe for owner fencing because each resident `CellHandle` rejects drained ownership; it bounds fresh authority discovery at 500 ms when the cache is enabled. +Increasing the local handle cache from 50 ms to 500 ms reduces repeated authority resolution inside a transaction. On a fresh four-partition fixture, signed `TransactWriteItems` reached 4.46 requests/s at one client (p95 306 ms) and 1.55 requests/s at four clients (p95 2.87 s), with zero errors. An eight-client run reached 2.02 requests/s before one `ServiceUnavailable`; the coordinator and durable participant Cells still contend under concurrency. The longer cache remains safe for owner fencing because each resident `CellHandle` rejects drained ownership. This historical sample predates the resident-actor resolution path. The same earlier run measured `Scan` at 1,101.1/1,660.0 requests/s, `BatchGetItem` at 857.2/1,498.2 requests/s (1,714.3/2,996.4 items/s), `BatchWriteItem` at 33.8/64.3 requests/s (67.5/128.5 items/s), `DescribeTable` at 919.0/1,474.0 requests/s, and `ListTables` at 1,457.8/1,725.0 requests/s for one/eight clients. Transactions reached 4.17/4.95 `TransactGetItems` requests/s and 1.27/1.53 `TransactWriteItems` requests/s; all cases completed without request errors, but transaction latency and durable-write throughput remain the limiting gap. @@ -141,6 +1087,69 @@ The pinned ExtendDB `BatchWriteItem` handler awaits each item mutation in a requ Temporary stage timings on a separate instrumented fixture put typical credential, IAM, table record, and data Cell reads around 3–4 ms each, while each route traversal took around 6–7 ms. The requests perform several of these operations in sequence. The instrumented write fixture differed materially from the clean fixture, so its write timings are not a publication-cost estimate. A temporary S3 proxy disrupted publication and its counts were discarded. +## Observe runtime costs and durability acknowledgements + +For a diagnostic run, add an absolute path to the server's JSON configuration: + +```json +{ + "runtime_metrics_file": "/var/lib/beyonddb/runtime-metrics.json" +} +``` + +The parent directory must exist. Give each process its own file path. The server +refreshes the file once per second through a temporary file and atomic rename. +This option is disabled by default; enabling it adds atomic counter updates and +a periodic snapshot task. A write failure logs a warning and sampling continues. + +| Observation | What it measures | +| --- | --- | +| `command_responses` | Runtime command replies using recorded results, follower proof (`fleet`), or object publication (`object`) | +| `command_queue`, `command_worker` | Command admission queue and worker execution time | +| `primitive_execution` | Command and query primitive execution time | +| `publication` | Queue, preparation, authority, total duration, and uploaded objects/bytes | +| `activation` | Ownership, resume, root opening, restore, and activation phases | +| `durability_submissions`, `follower_appends` | Follower submission outcomes and append acknowledgement/failure counts | +| `follower_append_phases` | Sender peer lookup and HTTP round trip; receiver fresh enrollment and durable store append | +| `catalog_reads`, `control_reads` | Reads observed by the runtime telemetry hooks | +| `node_resources` | Sampled active Cells, retained bytes, worker jobs, and unpublished log bytes | +| `object_store` | Logical storage operations across the configured provider, including routing and enrollment I/O | + +Timing objects contain cumulative `count`, `failed`, `total_us`, and `max_us`. +Durations use microseconds. Counter snapshots are approximate because work can +continue during sampling. They include background commands; runtime command +reply counts are **not SDK request counts** and exclude queries, transport, and +abandoned replies. Catalog/control counters do not cover all application or +object-store reads. Application routing, peer metadata, and fresh bootstrap +provisioning can bypass these hooks: zero activation/catalog/control counters +do not imply zero work in those phases. These counters cannot provide p95 latency. + +`object_store` adds fixed operation and outcome labels, byte totals, `started`, +and `in_flight` counts. It observes the shared provider used by the server, peer +directory, and provisioner. A read finishes when its body is consumed or dropped; +duration includes that lifetime, rather than measuring network time alone. +`failed` includes normal `not_found`, `conflict`, and `cancelled` outcomes, so it +is **not an SDK error count**. No object paths, tenant IDs, or credentials are +recorded. These counters also include startup and background work. + +`follower_append_phases` observes Append only. Seal, Tail and Retire are excluded. +The sender's `round_trip` starts after peer lookup and includes the complete +bounded HTTP response body. The receiver's `enrollment` covers the fresh +mTLS-bound canonical read; `durable_append` covers `FollowerStore::append`, +including its durable acknowledgement. Neither receiver phase includes HTTP +body admission or response scheduling. Timings add `started`, `in_flight` and +`cancelled`; dropping an active future records a cancelled failure. They use +fixed labels with no node/session IDs. Sender and receiver observations cover +different, overlapping scopes: do not subtract their aggregate means to claim +network latency or add them into SDK latency. + +Save snapshots before and after a run, check that +`first_snapshot_at_unix_ms` is unchanged, and check the freshness of +`sampled_at_unix_ms`. A server restart begins a new counter series. Divide the +change in `total_us` by the change in `count` to estimate a phase's mean duration. +Phases can overlap and cover different events; do not sum them into SDK latency. +Use the signed SDK harness for end-to-end latency and foreground errors. + ## Run a repeatable point-operation sample Build an optimized server with `cargo build --release`, start it using [the deployment guide](deployment.md), and create a dedicated table. The commands below use test credentials already configured for that server. @@ -181,6 +1190,8 @@ The script seeds 64 one-KiB items, signs requests through boto3, disables SDK re ## Interpret the result +The current [harness](../scripts/bench.py) sorts **successful** request durations and selects sample index `floor(p × (n - 1))`, then rounds to two decimals. It does not interpolate. With two successes, the reported p95 is the smaller duration; with eight, it is the second largest. Always read percentiles together with completion counts, actual elapsed time and errors. Low-count transaction cases do not establish a production tail-latency target. + Short, closed-loop runs are useful for comparing code changes under the *same* fixture. They are not a peak TPS rating. Report at least the build profile and Git revision; server and client CPU; object-store latency and request counts; table partitions; item size and key distribution; Streams/index settings; client concurrency; errors; p99; and recovery after owner loss. Test hot keys separately from evenly spread keys. Run sustained mixed read/write load, table creation, splits, transactions, and restarts before using a result for capacity planning. A healthy `/health` response alone does not prove that background work is keeping up. Do not use a debug build or a node reporting mailbox exhaustion to set a capacity target. Fix readiness and background-work errors before comparing throughput. diff --git a/docs/scaling.md b/docs/scaling.md index 45dc21a..d01f6ba 100644 --- a/docs/scaling.md +++ b/docs/scaling.md @@ -44,6 +44,39 @@ and a durable projection journal. Bounded tombstone retention, projection throughput, and index-owner fleet recovery remain open scale gates. See [global indexes](global-indexes.md). +## Choose a table's Cell model + +Choose placement when you create a table with the `beyonddb:cell-model` tag. This BeyondDB extension uses the standard DynamoDB `CreateTable.Tags` field, so AWS CLI and SDK clients need no custom request format. + +| Value | Initial base data Cells | Base splitting | Use when | +| --- | ---: | --- | --- | +| `single` | 1 | Disabled, including manual splits | Your table fits one Cell's disk, memory, write, and recovery budgets. | +| `auto` | 1 | Enabled | You want to start with one Cell and allow range growth. | +| `partitioned` | `max(2, initial_partitions)` | Enabled | You need independent ranges from creation. | +| Tag omitted | Configured `initial_partitions` | Enabled | You want to preserve the existing server default. | + +```text +single auto / partitioned +Table directory Table directory + | / \ +One base data Cell Data Cell A Data Cell B +items + LSIs + journal hash range A hash range B + | \ / + +---- async projection ------> GSI Cells +``` + +`single` means one **base data Cell**. Account metadata, credentials, directories, and transaction coordinators remain separate. Each global secondary index (GSI) also uses separate Cells; a single-model table starts each GSI with one range, and the index can split independently. Local secondary indexes (LSIs) stay with base items. + +The choice persists in the table generation. Changing or removing the tag later changes tag metadata only; `TagResource`, `UntagResource`, and `UpdateTable` do not move data or change placement. Live conversion from `single` to a splitting model is not implemented. Choose `auto` if future growth is uncertain. Invalid values and duplicate selector tags are rejected before table creation. + +Table size and write load both matter. A smaller table with heavy writes may need several Cells. One Cell serializes mutations and remains subject to node admission, disk budgets, and recovery costs. `single` suppresses splitting; it does not remove resource limits. Hash-range splitting also does not solve every hot partition-key workload; see the delivery gates below. + +Start with `auto` when you want one base data Cell today and room to split later. Choose `single` when fixed placement matters and you can keep the complete Cell within its resource budget. Choose `partitioned` when measured write load needs several independent owners from the beginning. Labels such as “medium” or “large” do not determine placement: both stored bytes and write demand matter. + +**Current sizing limit:** the compiled base data Cell has a **512 MiB SQLite database budget** and a **64 MiB capture budget**. The database budget includes items, local indexes, stream records, transaction state and other Cell metadata; usable item storage is smaller. The creation tag does not raise these limits. A table that will exceed this budget needs `auto` or `partitioned` in the current release. Configurable larger single-Cell budgets and live placement conversion are unfinished. + +Small same-Cell transaction reads can use one atomic query. Tokenized transaction writes still require the account-scoped coordinator, and cross-table transactions can span Cells. Placement therefore reduces some transaction work without establishing SQLite write parity. See [transaction execution paths](cross-cell-transactions.md#one-cell-and-multiple-cell-requests) and [measured performance](performance.md). + ## Horizontal scaling delivery gates The agreed target is 10,000 active Cells and multi-TB stored data. The following diff --git a/docs/user-guide.md b/docs/user-guide.md index 5235cd9..8a0b98d 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -33,6 +33,7 @@ Create `Notes`, then wait for its route to become active before writing. This ex ```sh aws dynamodb create-table \ --table-name Notes \ + --tags Key=beyonddb:cell-model,Value=single \ --attribute-definitions AttributeName=pk,AttributeType=S \ --key-schema AttributeName=pk,KeyType=HASH \ --billing-mode PAY_PER_REQUEST \ @@ -47,7 +48,18 @@ aws dynamodb describe-table \ --endpoint-url "$BEYONDDB_ENDPOINT" ``` -The serving binary provisions `initial_partitions` data Cells for a new routed table. `DescribeTable` can report `CREATING` until range publication finishes. Changing `initial_partitions` later affects new table generations only. The node's `max_active_cells` budget must include those data Cells plus account, coordinator, and management Cells; if the budget is too small, provisioning remains pending until capacity is available. `sql_workers` is an optional override for the SQL worker count (maximum sixteen); the default follows host parallelism. Use it for multi-partition workloads after measuring CPU and memory headroom. It improves independent Cell scheduling, while a single hot Cell remains serialized for ordering and durable publication. +This example keeps the table in one base data Cell. Choose a model at creation: + +| `beyonddb:cell-model` value | Behavior | +| --- | --- | +| `single` | One base data Cell; no base splits | +| `auto` | Start with one base data Cell; allow splits | +| `partitioned` | Start with at least two base data Cells, using the configured count | +| Omitted | Use the server's existing `initial_partitions` default | + +The selector is a BeyondDB extension. Changing tags after creation does not change placement. GSIs remain separate and can grow independently; LSIs share base storage. Live model conversion is unfinished. Read [Cell model selection and limits](scaling.md#choose-a-tables-cell-model) before choosing a fixed single Cell. + +Without the selector, the serving binary provisions `initial_partitions` data Cells for a new routed table. `DescribeTable` can report `CREATING` until range publication finishes. Changing `initial_partitions` later affects new table generations only. The node's `max_active_cells` budget must include those data Cells plus account, coordinator, and management Cells; if the budget is too small, provisioning remains pending until capacity is available. `sql_workers` is an optional override for the SQL worker count (maximum sixteen); the default follows host parallelism. Use it for multi-partition workloads after measuring CPU and memory headroom. It improves independent Cell scheduling, while a single hot Cell remains serialized for ordering and durable publication. ## Write and read an item diff --git a/src/backend.rs b/src/backend.rs index 533dd46..9d2c530 100644 --- a/src/backend.rs +++ b/src/backend.rs @@ -5,6 +5,7 @@ mod batch; mod data; mod global_index; mod metadata_cache; +mod prepare_batch; mod recovery; mod remaining; mod statistics; @@ -14,6 +15,7 @@ pub(crate) mod table_creation; mod transaction; mod transaction_read; mod transaction_transport; +mod update_batch; use std::{ collections::{HashMap, HashSet}, @@ -42,6 +44,8 @@ use extenddb_storage::{BoxedFuture, TableEngine}; use batch::NoReturnBatcher; use metadata_cache::MetadataCache; +use prepare_batch::PrepareBatcher; +use update_batch::UpdateBatcher; /// Installs an initial table's data Cells before its route becomes visible. pub trait InitialPartitionProvisioner: Send + Sync { @@ -90,6 +94,8 @@ pub trait CoordinatorProvisioner: Send + Sync { pub struct CellStorage { client: CellClient, no_return_batcher: Arc, + update_batcher: Arc, + prepare_batcher: Arc, region: String, initial_partitions: Option>, coordinators: Option>, @@ -126,6 +132,8 @@ impl CellStorage { let client = client.with_read_policy(ReadPolicy::CurrentOwner); Self { no_return_batcher: Arc::new(NoReturnBatcher::new(client.clone())), + update_batcher: Arc::new(UpdateBatcher::new(client.clone())), + prepare_batcher: Arc::new(PrepareBatcher::new(client.clone())), client, region: region.into(), initial_partitions: None, @@ -331,12 +339,12 @@ impl TableEngine for CellStorage { let name = input.table_name.clone(); let spec = TableSpec { table_class: table_class(input.table_class.as_deref())?.unwrap_or_default(), - placement: self.initial_partitions.as_ref().map_or( - TablePlacement::Account, - |provisioner| TablePlacement::Routed { - initial_partitions: provisioner.initial_partition_count(), - }, - ), + placement: table_creation::placement( + self.initial_partitions + .as_ref() + .map(|provisioner| provisioner.initial_partition_count()), + input.tags.as_deref().unwrap_or_default(), + )?, table_name: input.table_name, key_schema: input.key_schema, attribute_definitions: input.attribute_definitions, @@ -353,6 +361,10 @@ impl TableEngine for CellStorage { )), stream, }; + let explicit_model = spec + .initial_tags + .iter() + .any(|tag| tag.key == table_creation::CELL_MODEL_TAG); let submitted = spec.clone(); let record = match self .client @@ -378,6 +390,7 @@ impl TableEngine for CellStorage { return Err(StorageError::TableAlreadyExists(name)); }; if existing.placement == TablePlacement::Account + || (explicit_model && submitted.placement != existing.placement) || !submitted.matches_record(&existing) || self.route_active_for(&account_id, &existing.id).await? { @@ -506,7 +519,7 @@ impl TableEngine for CellStorage { } crate::TableLifecycle::Deleting(record) => (record, TableStatus::Deleting), crate::TableLifecycle::Live(record) => { - let status = if matches!(record.placement, TablePlacement::Routed { .. }) + let status = if record.placement.is_routed() && !self.route_active_for(&account_id, &record.id).await? { TableStatus::Creating @@ -669,13 +682,24 @@ impl TableEngine for CellStorage { let account_id = account_id.to_owned(); let table_name = table_name.to_owned(); Box::pin(async move { + // Batch APIs can bypass ExtendDB's HTTP table-info cache. Reuse + // the same opt-in backend policy; item access still checks table ID + // and partition epoch, so a stale entry cannot alias a new table. + if self.route_cache_enabled + && let Some(cached) = self.metadata_cache.key_info(&account_id, &table_name) + { + return Ok(cached); + } + let generation = self + .route_cache_enabled + .then(|| self.metadata_cache.generation()); let record = self.record(&account_id, &table_name).await?; - if matches!(record.placement, TablePlacement::Routed { .. }) + if record.placement.is_routed() && !self.route_active_for(&account_id, &record.id).await? { return Err(StorageError::TableNotActive(table_name)); } - Ok(TableKeyInfo { + let info = TableKeyInfo { has_lsi: !record.local_secondary_indexes.is_empty(), local_secondary_indexes: record .local_secondary_indexes @@ -698,7 +722,16 @@ impl TableEngine for CellStorage { key_schema: record.key_schema, attribute_definitions: record.attribute_definitions, ..TableKeyInfo::default() - }) + }; + if let Some(generation) = generation { + self.metadata_cache.insert_key_info( + &info.account_id, + table_name, + generation, + info.clone(), + ); + } + Ok(info) }) } diff --git a/src/backend/admission.rs b/src/backend/admission.rs index 262ff50..67e9009 100644 --- a/src/backend/admission.rs +++ b/src/backend/admission.rs @@ -17,11 +17,15 @@ use crate::{ TransactionOperation, TransactionToken, coordinator_target, }; +const ACKNOWLEDGED_BEGIN_BYTES: usize = 32 * 1024; + pub(super) struct AdmittedTransaction { pub identity: ReadCrossCellTransactionInput, pub decision: CoordinatorDecision, pub participant_count: u8, pub replay: bool, + /// Exact bounded BEGIN payload, retained only for fresh read assembly. + pub acknowledged_read_participants: Option>, } impl CellStorage { @@ -156,10 +160,15 @@ impl CellStorage { participants: participants.into_values().collect(), }; let identity = mutation_identity()?; - let inline = serde_json::to_vec(&input) + let input_bytes = serde_json::to_vec(&input) .map_err(|error| StorageError::Internal(error.to_string()))? - .len() - <= crate::transaction_transport::INLINE_BYTES; + .len(); + let inline = input_bytes <= crate::transaction_transport::INLINE_BYTES; + // Retain only a bounded first-attempt payload. BEGIN's acknowledged + // Begun outcome proves that this exact ordered participant set committed. + // Existing identities and ambiguous replies must read durable state. + let acknowledged_participants = + (input_bytes <= ACKNOWLEDGED_BEGIN_BYTES).then(|| input.participants.clone()); let result = if inline { self.client .command::( @@ -183,6 +192,34 @@ impl CellStorage { let (transaction_id, prior) = match result { Ok(result) => match result.output.0 { BeginCrossCellTransactionOutcome::Begun => { + if let Some(participants) = acknowledged_participants { + let acknowledged_read_participants = participants + .iter() + .all(|participant| { + matches!( + participant.target, + CoordinatorParticipantTarget::Data { .. } + ) && participant.operations.iter().all(|operation| { + matches!(operation.operation, TransactionOperation::Read(_)) + }) + }) + .then(|| participants.clone()); + let identity = ReadCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id: proposed_id, + routing_key, + }; + let status = self + .drive_acknowledged_transaction(&coordinator, &identity, participants) + .await?; + return Ok(AdmittedTransaction { + identity, + decision: status.decision, + participant_count: status.participant_count, + replay: false, + acknowledged_read_participants, + }); + } (proposed_id, CoordinatorDecision::Begin) } BeginCrossCellTransactionOutcome::Existing { @@ -255,6 +292,7 @@ impl CellStorage { decision: status.decision, participant_count: status.participant_count, replay: prior == CoordinatorDecision::Commit, + acknowledged_read_participants: None, }) } diff --git a/src/backend/batch.rs b/src/backend/batch.rs index e7c6913..38afffb 100644 --- a/src/backend/batch.rs +++ b/src/backend/batch.rs @@ -1,4 +1,4 @@ -//! Bounded coalescing for idempotent routed mutations. +//! Coalescing for routed mutations that need no returned item images. use std::collections::{HashMap, VecDeque}; use std::sync::{ @@ -21,7 +21,7 @@ use crate::{ const BATCH_WINDOW: Duration = Duration::from_millis(2); const MAX_BATCH_OPERATIONS: usize = 16; -/// An unconditional mutation that can be replayed safely as part of a batch. +/// An unconditional mutation committed with the batch's mutation identity. #[derive(Clone)] pub(crate) struct NoReturnMutation { /// Table generation checked by the partition command. @@ -45,7 +45,7 @@ struct Slot { scheduled: AtomicBool, } -/// Coalesces a small number of idempotent writes before one durable command. +/// Coalesces independent writes before one durable command. pub(crate) struct NoReturnBatcher { client: cellule_runtime::CellClient, slots: Mutex>>, diff --git a/src/backend/data.rs b/src/backend/data.rs index 8ac8b24..930e2ca 100644 --- a/src/backend/data.rs +++ b/src/backend/data.rs @@ -35,12 +35,11 @@ use crate::{ PartitionPutInput, PartitionPutOutcome, PartitionQuery, PartitionQueryInput, PartitionQueryOutcome, PartitionTransactReadOutcome, PartitionTransactReadQuery, PartitionTransactWrite, PartitionTransactWriteInput, PartitionTransactWriteNoReturn, - PartitionTransactWriteOutcome, PartitionUpdate, PartitionUpdateInput, PartitionUpdateOutcome, - PutItem, PutItemInput, PutItemNoReturn, ScanItems, ScanItemsInput, ScanItemsOutcome, - SortComparison, SortPredicate, TransactReadQuery, TransactWrite, TransactWriteInput, - TransactWriteNoReturn, TransactionFailure, TransactionOperation, TransactionOutcome, - TransactionReadOutcome, UpdateItem, UpdateItemInput, UpdateItemNoReturn, UpdateItemOutcome, - data_key_hash, + PartitionTransactWriteOutcome, PartitionUpdateInput, PartitionUpdateOutcome, PutItem, + PutItemInput, PutItemNoReturn, ScanItems, ScanItemsInput, ScanItemsOutcome, SortComparison, + SortPredicate, TransactReadQuery, TransactWrite, TransactWriteInput, TransactWriteNoReturn, + TransactionFailure, TransactionOperation, TransactionOutcome, TransactionReadOutcome, + UpdateItem, UpdateItemInput, UpdateItemNoReturn, UpdateItemOutcome, data_key_hash, }; use cellule_runtime::client::InvocationError; use cellule_runtime::identity::CellTarget; @@ -101,6 +100,7 @@ impl CellStorage { match result { Ok(committed) => match committed.output.0 { TransactionReadOutcome::Applied(items) => Ok(Some(items)), + TransactionReadOutcome::SavedImagesRequired => Ok(None), TransactionReadOutcome::Rejected { index, reason } => { Err(transaction_canceled(index, reason, count, &[])) } @@ -565,45 +565,24 @@ impl DataEngine for CellStorage { update, condition, }; - let (old, new) = if return_old || return_new { - match self - .client - .command::(&partition, mutation_identity()?, Json(input)) - .await - { - Ok(committed) => match committed.output.0 { - PartitionUpdateOutcome::Applied { old, new } => (old, new), - _ => { - return Err(StorageError::Internal( - "unexpected partition update".into(), - )); - } - }, - Err(InvocationError::Rejected(committed)) => { - self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); - return Err(partition_update_rejection(committed.output.0)); - } - Err(error) => return Err(cell_error(error)), - } - } else { - match self - .client - .command::(&partition, mutation_identity()?, Json(input)) - .await - { - Ok(committed) => match committed.output.0 { - PartitionUpdateOutcome::Applied { old, new } => (old, new), - _ => { - return Err(StorageError::Internal( - "unexpected partition update result".into(), - )); - } + let dedup_key = crate::item_key(&input.key, &key_info.base_key_schema) + .map_err(|error| StorageError::Internal(error.to_string()))?; + let outcome = self + .submit_update( + partition, + dedup_key, + crate::BatchedPartitionUpdate { + input, + return_old, + return_new, }, - Err(InvocationError::Rejected(committed)) => { - self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); - return Err(partition_update_rejection(committed.output.0)); - } - Err(error) => return Err(cell_error(error)), + ) + .await?; + let (old, new) = match outcome { + PartitionUpdateOutcome::Applied { old, new } => (old, new), + other => { + self.invalidate_route_cache(&key_info.account_id, &key_info.table_id); + return Err(partition_update_rejection(other)); } }; return Ok(( @@ -1275,6 +1254,7 @@ fn local_partition_transaction_read_result( ) -> Result>>, StorageError> { match outcome { PartitionTransactReadOutcome::Applied(items) => Ok(Some(items)), + PartitionTransactReadOutcome::SavedImagesRequired => Ok(None), PartitionTransactReadOutcome::Rejected { index, reason } => { Err(transaction_canceled(index, reason, count, &[])) } diff --git a/src/backend/global_index.rs b/src/backend/global_index.rs index 40ab8ba..59f5ad9 100644 --- a/src/backend/global_index.rs +++ b/src/backend/global_index.rs @@ -425,7 +425,7 @@ impl CellStorage { }); let account_target = target(account)?; let page = async { - if matches!(table.placement, crate::TablePlacement::Routed { .. }) { + if table.placement.is_routed() { provisioner .recover_route_directory_path( &self.client, diff --git a/src/backend/metadata_cache.rs b/src/backend/metadata_cache.rs index 22db02f..1c3e1f9 100644 --- a/src/backend/metadata_cache.rs +++ b/src/backend/metadata_cache.rs @@ -6,7 +6,7 @@ use std::{ time::{Duration, Instant}, }; -use extenddb_core::types::{ListTablesOutput, TableDescription}; +use extenddb_core::types::{ListTablesOutput, TableDescription, TableKeyInfo}; const TTL: Duration = Duration::from_millis(500); const MAX_ENTRIES: usize = 128; @@ -21,6 +21,7 @@ struct State { generation: u64, descriptions: HashMap<(String, String), Entry>, listings: HashMap<(String, i64, Option), Entry>, + key_infos: HashMap<(String, String), Entry>, } struct Entry { @@ -48,6 +49,39 @@ impl MetadataCache { self.read().generation } + pub(super) fn key_info(&self, account_id: &str, name: &str) -> Option { + let state = self.read(); + let entry = state + .key_infos + .get(&(account_id.to_owned(), name.to_owned()))?; + (entry.at.elapsed() < TTL && entry.generation == state.generation) + .then(|| entry.value.clone()) + } + + pub(super) fn insert_key_info( + &self, + account_id: &str, + name: String, + generation: u64, + value: TableKeyInfo, + ) { + let mut state = self.write(); + if state.generation != generation { + return; + } + if state.key_infos.len() >= MAX_ENTRIES { + state.key_infos.clear(); + } + state.key_infos.insert( + (account_id.to_owned(), name), + Entry { + at: Instant::now(), + generation, + value, + }, + ); + } + pub(super) fn description(&self, account_id: &str, name: &str) -> Option { let state = self.read(); let entry = state @@ -125,5 +159,6 @@ impl MetadataCache { state.generation = state.generation.saturating_add(1); state.descriptions.clear(); state.listings.clear(); + state.key_infos.clear(); } } diff --git a/src/backend/prepare_batch.rs b/src/backend/prepare_batch.rs new file mode 100644 index 0000000..818d464 --- /dev/null +++ b/src/backend/prepare_batch.rs @@ -0,0 +1,318 @@ +//! Bounded publication sharing for independent routed transaction prepares. + +use std::collections::{HashMap, VecDeque}; +use std::sync::{ + Arc, Weak, + atomic::{AtomicBool, Ordering}, +}; +use std::time::Duration; + +use cellule_runtime::client::{CellClient, Committed, InvocationError, Receipt}; +use cellule_runtime::identity::{CellTarget, RequestId}; +use extenddb_storage::error::StorageError; +use tokio::sync::{Mutex, OwnedSemaphorePermit, Semaphore, oneshot}; + +use super::transaction_transport::PhaseError; +use super::{cell_error, mutation_identity}; +use crate::partition::{MAX_PREPARES, PREPARE_BATCH_BYTES}; +use crate::{ + Json, ParticipantTransactionState, PreparePartitionBatchOutcome, PreparePartitionTransaction, + PreparePartitionTransactionBatch, PreparePartitionTransactionBounded, + PreparePartitionTransactionInput, PrepareTransactionOutcome, ReadPartitionTransaction, + ReadTransactionInput, TransactionCommandInput, +}; + +type Prepared = Result<(PrepareTransactionOutcome, Receipt), PhaseError>; +const WINDOW: Duration = Duration::from_millis(2); + +struct Pending { + input: PreparePartitionTransactionInput, + bytes: usize, + reply: oneshot::Sender, + _permit: OwnedSemaphorePermit, +} + +struct Slot { + target: CellTarget, + queued: Mutex>, + permits: Arc, + scheduled: AtomicBool, +} + +pub(super) struct PrepareBatcher { + client: CellClient, + slots: Mutex>>, +} + +impl PrepareBatcher { + pub(super) fn new(client: CellClient) -> Self { + Self { + client, + slots: Mutex::new(HashMap::new()), + } + } + + pub(super) async fn submit( + self: &Arc, + target: CellTarget, + input: PreparePartitionTransactionInput, + ) -> Prepared { + let bytes = serde_json::to_vec(&input) + .map_err(|error| StorageError::Internal(error.to_string()))? + .len() + + 1; + if bytes + 2 > PREPARE_BATCH_BYTES { + return self.individual(&target, input).await; + } + let slot = { + let mut slots = self.slots.lock().await; + let id = *target.cell_id().as_bytes(); + if let Some(slot) = slots.get(&id).and_then(Weak::upgrade) { + slot + } else { + slots.retain(|_, slot| slot.strong_count() > 0); + let slot = Arc::new(Slot { + target, + queued: Mutex::new(VecDeque::new()), + permits: Arc::new(Semaphore::new(64)), + scheduled: AtomicBool::new(false), + }); + slots.insert(id, Arc::downgrade(&slot)); + slot + } + }; + let permit = slot + .permits + .clone() + .acquire_owned() + .await + .map_err(|_| StorageError::Transient("prepare queue closed".into()))?; + let (reply, result) = oneshot::channel(); + let schedule = { + let mut queued = slot.queued.lock().await; + queued.push_back(Pending { + input, + bytes, + reply, + _permit: permit, + }); + !slot.scheduled.swap(true, Ordering::AcqRel) + }; + if schedule { + let batcher = self.clone(); + tokio::spawn(async move { + batcher.flush(slot).await; + }); + } + result + .await + .map_err(|_| StorageError::Transient("prepare worker stopped".into()))? + } + + async fn flush(self: Arc, slot: Arc) { + loop { + if slot.queued.lock().await.len() < MAX_PREPARES { + tokio::time::sleep(WINDOW).await; + } + let pending = { + let mut queue = slot.queued.lock().await; + queue.retain(|pending| !pending.reply.is_closed()); + if queue.is_empty() { + slot.scheduled.store(false, Ordering::Release); + return; + } + let mut selected = Vec::new(); + let mut bytes = 2; + while selected.len() < MAX_PREPARES { + let Some(candidate) = queue.front() else { + break; + }; + if bytes + candidate.bytes > PREPARE_BATCH_BYTES { + break; + } + let Some(candidate) = queue.pop_front() else { + break; + }; + bytes += candidate.bytes; + selected.push(candidate); + } + selected + }; + if pending.len() == 1 { + for pending in pending { + let result = self.individual(&slot.target, pending.input).await; + let _ = pending.reply.send(result); + } + continue; + } + let identity = match mutation_identity() { + Ok(identity) => identity, + Err(error) => { + for pending in pending { + let _ = pending.reply.send(Err(error.clone().into())); + } + continue; + } + }; + let inputs = pending + .iter() + .map(|pending| pending.input.clone()) + .collect(); + let result = self + .client + .command::(&slot.target, identity, Json(inputs)) + .await; + match result { + Ok(committed) if committed.output.0 == PreparePartitionBatchOutcome::Prepared => { + for pending in pending { + let _ = pending + .reply + .send(Ok((PrepareTransactionOutcome::Prepared, committed.receipt))); + } + } + Err(InvocationError::Rejected(committed)) + if committed.output.0 == PreparePartitionBatchOutcome::IndividualRequired => + { + // This receipt proves rollback of every application effect. + // Never use this fallback for an uncertain batch outcome. + for pending in pending { + if !pending.reply.is_closed() { + let result = self.individual(&slot.target, pending.input).await; + let _ = pending.reply.send(result); + } + } + } + Err(InvocationError::Pending(_)) => { + for pending in pending { + let result = self.observe(&slot.target, &pending.input).await; + let _ = pending.reply.send(result); + } + } + Err(InvocationError::NotStarted(cellule_runtime::Error::Capacity(reason))) => { + // The complete batch is proven not to have started. Its + // combined capture/resolution reservation may exceed the + // budget even when each participant fits independently. + tracing::debug!( + reason, + "prepare batch capacity requires individual commands" + ); + for pending in pending { + if !pending.reply.is_closed() { + let result = self.individual(&slot.target, pending.input).await; + let _ = pending.reply.send(result); + } + } + } + Err(InvocationError::NotStarted(cellule_runtime::Error::Sqlite(error))) + if error.sqlite_error_code() + == Some(cellule_ltx::rusqlite::ErrorCode::DiskFull) => + { + // Preserve individual FULL admission classification so + // the driver can durably decide ABORT when needed. + for pending in pending { + if !pending.reply.is_closed() { + let result = self.individual(&slot.target, pending.input).await; + let _ = pending.reply.send(result); + } + } + } + Err(error) => { + let error = cell_error(error); + for pending in pending { + let _ = pending.reply.send(Err(error.clone().into())); + } + } + Ok(_) => { + for pending in pending { + let _ = pending.reply.send(Err(StorageError::Internal( + "invalid prepare batch reply".into(), + ) + .into())); + } + } + } + } + } + + async fn individual( + &self, + target: &CellTarget, + input: PreparePartitionTransactionInput, + ) -> Prepared { + let identity = mutation_identity()?; + let result = self + .client + .command::(target, identity, Json(input.clone())) + .await; + let result = if matches!(&result, Err(InvocationError::Rejected(committed)) + if committed.output.0 == PrepareTransactionOutcome::WideRequired) + { + // Different command digests require different mutation identities. + let identity = cellule_runtime::MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + ..identity + }; + self.client + .command::( + target, + identity, + Json(TransactionCommandInput::Inline(input.clone())), + ) + .await + } else { + result + }; + self.finish_individual(target, &input, result).await + } + + async fn finish_individual( + &self, + target: &CellTarget, + input: &PreparePartitionTransactionInput, + result: Result< + Committed>, + InvocationError>, + >, + ) -> Prepared { + match result { + Ok(committed) => Ok((committed.output.0, committed.receipt)), + Err(InvocationError::Rejected(committed)) => { + Ok((committed.output.0, committed.receipt)) + } + Err(InvocationError::Pending(_)) => self.observe(target, input).await, + Err(error) => Err(error.into()), + } + } + + async fn observe( + &self, + target: &CellTarget, + input: &PreparePartitionTransactionInput, + ) -> Prepared { + let observed = self + .client + .query::( + target, + None, + Json(ReadTransactionInput { + transaction_id: input.transaction_id, + coordinator_cell: input.coordinator_cell, + }), + ) + .await + .map_err(cell_error)?; + let outcome = match observed.output.0 { + ParticipantTransactionState::Prepared => PrepareTransactionOutcome::Replay, + ParticipantTransactionState::Committed => PrepareTransactionOutcome::Committed, + ParticipantTransactionState::Aborted => PrepareTransactionOutcome::Aborted, + ParticipantTransactionState::CoordinatorMismatch => PrepareTransactionOutcome::Mismatch, + ParticipantTransactionState::Missing => { + return Err(StorageError::Transient( + "participant prepare outcome remains pending".into(), + ) + .into()); + } + }; + Ok((outcome, observed.receipt)) + } +} diff --git a/src/backend/recovery.rs b/src/backend/recovery.rs index d2d1467..e2733e9 100644 --- a/src/backend/recovery.rs +++ b/src/backend/recovery.rs @@ -204,6 +204,36 @@ impl CellStorage { .map_err(cell_error)? .output .0; + self.finish_transaction_participants(coordinator, read, commit, participants) + .await?; + let final_status = self + .client + .query::(coordinator, None, Json(read.clone())) + .await + .map_err(cell_error)? + .output + .0 + .ok_or_else(|| StorageError::Internal("coordinator transaction disappeared".into()))?; + if final_status.resolved_count != final_status.participant_count + || final_status.unreleased_read_results != 0 + { + return Err(StorageError::Transient( + "cross-Cell participant resolution is incomplete".into(), + )); + } + Ok(()) + } + + /// Publish each participant's durable resolution receipt. Success requires + /// an acknowledged coordinator record for every supplied participant. + /// The caller must supply an authoritative terminal decision and targets. + pub(super) async fn finish_transaction_participants( + &self, + coordinator: &CellTarget, + read: &ReadCrossCellTransactionInput, + commit: bool, + participants: Vec, + ) -> Result<(), StorageError> { let mut failure = None; // Only terminal decisions permit independent resolution. Keep a small // window so a slow owner cannot hold healthy keys, without fanning one @@ -255,25 +285,7 @@ impl CellStorage { } self.record_resolution_progress(coordinator, read, &mut read_releases, &mut resolutions) .await?; - if let Some(error) = failure { - return Err(error); - } - let final_status = self - .client - .query::(coordinator, None, Json(read.clone())) - .await - .map_err(cell_error)? - .output - .0 - .ok_or_else(|| StorageError::Internal("coordinator transaction disappeared".into()))?; - if final_status.resolved_count != final_status.participant_count - || final_status.unreleased_read_results != 0 - { - return Err(StorageError::Transient( - "cross-Cell participant resolution is incomplete".into(), - )); - } - Ok(()) + failure.map_or(Ok(()), Err) } async fn record_resolution_progress( @@ -318,6 +330,7 @@ impl CellStorage { } } if !resolutions.is_empty() { + let expected = resolutions.len(); let recorded = self .client .command::( @@ -327,12 +340,14 @@ impl CellStorage { ) .await .map_err(cell_error)?; - if recorded.output.0.iter().any(|outcome| { - !matches!( - outcome, - CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay - ) - }) { + if recorded.output.0.len() != expected + || recorded.output.0.iter().any(|outcome| { + !matches!( + outcome, + CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay + ) + }) + { return Err(StorageError::Internal( "coordinator rejected participant resolution".into(), )); diff --git a/src/backend/statistics.rs b/src/backend/statistics.rs index 46eb57f..2876269 100644 --- a/src/backend/statistics.rs +++ b/src/backend/statistics.rs @@ -31,7 +31,7 @@ impl Sweep { snapshot: StatisticsSnapshot { table_id: table.id.clone(), sampled_at: mutation_identity()?.issued_at_ms, - base_routed: matches!(table.placement, crate::TablePlacement::Routed { .. }), + base_routed: table.placement.is_routed(), index_generations: Default::default(), statistics: TableStatistics::default(), }, diff --git a/src/backend/streams.rs b/src/backend/streams.rs index a62b631..c4c7ad3 100644 --- a/src/backend/streams.rs +++ b/src/backend/streams.rs @@ -148,7 +148,7 @@ impl CellStorage { } _ => StreamStatus::Disabled, } - } else if matches!(record.placement, crate::TablePlacement::Routed { .. }) + } else if record.placement.is_routed() && !self.route_active_for(account_id, &record.id).await? { StreamStatus::Enabling @@ -301,9 +301,9 @@ impl StreamEngine for CellStorage { let mut pending = if status == StreamStatus::Enabling { Vec::new() } else { - match record.placement { - crate::TablePlacement::Account => vec![(None, None)], - crate::TablePlacement::Routed { initial_partitions } => (0..initial_partitions) + match record.placement.initial_partitions() { + None => vec![(None, None)], + Some(initial_partitions) => (0..initial_partitions) .rev() .map(|index| { let mut partition_id = [0; 16]; @@ -532,7 +532,7 @@ impl StreamEngine for CellStorage { match (record.placement, shard) { (crate::TablePlacement::Account, StreamShard::Account { .. }) => Ok(()), ( - crate::TablePlacement::Routed { .. }, + crate::TablePlacement::Routed { .. } | crate::TablePlacement::Single, StreamShard::Partition { table_id, partition_id, diff --git a/src/backend/table_creation.rs b/src/backend/table_creation.rs index aef75e4..6ac4c9f 100644 --- a/src/backend/table_creation.rs +++ b/src/backend/table_creation.rs @@ -9,6 +9,47 @@ use crate::{ TableRecord, TableRoute, account_target, }; +pub(super) const CELL_MODEL_TAG: &str = "beyonddb:cell-model"; + +/// Interpret the opt-in creation tag at the protocol/backend boundary. +pub(super) fn placement( + configured_partitions: Option, + tags: &[extenddb_core::types::Tag], +) -> Result { + let mut models = tags.iter().filter(|tag| tag.key == CELL_MODEL_TAG); + let Some(model) = models.next() else { + return Ok( + configured_partitions.map_or(crate::TablePlacement::Account, |count| { + crate::TablePlacement::Routed { + initial_partitions: count, + } + }), + ); + }; + if models.next().is_some() { + return Err(StorageError::Validation( + "duplicate beyonddb:cell-model tag".into(), + )); + } + let Some(count) = configured_partitions else { + return Err(StorageError::Validation( + "cell model selection requires data Cell provisioning".into(), + )); + }; + match model.value.as_str() { + "single" => Ok(crate::TablePlacement::Single), + "auto" => Ok(crate::TablePlacement::Routed { + initial_partitions: 1, + }), + "partitioned" => Ok(crate::TablePlacement::Routed { + initial_partitions: count.max(2), + }), + _ => Err(StorageError::Validation( + "beyonddb:cell-model must be single, auto, or partitioned".into(), + )), + } +} + pub(crate) async fn publish_initial_routes( provisioner: &dyn InitialPartitionProvisioner, client: &CellClient, diff --git a/src/backend/transaction.rs b/src/backend/transaction.rs index e3fe27f..27066f1 100644 --- a/src/backend/transaction.rs +++ b/src/backend/transaction.rs @@ -9,14 +9,18 @@ use std::time::Duration; use super::transaction_transport::PhaseError; use super::{CellStorage, cell_error, mutation_identity}; use crate::{ - CoordinatorDecision, CoordinatorParticipantTarget, CoordinatorPhaseInput, - CoordinatorPhaseOutcome, CrossCellTransactionStatus, DecideCrossCellTransaction, - DecideCrossCellTransactionInput, DecideCrossCellTransactionOutcome, Json, - ParticipantTransactionState, PrepareAccountTransaction, PrepareAccountTransactionInput, - PreparePartitionTransaction, PreparePartitionTransactionInput, PrepareTransactionOutcome, - ReadCoordinatorParticipantInput, ReadCrossCellTransaction, ReadCrossCellTransactionInput, - ReadTransactionInput, ReadUnresolvedCoordinatorParticipants, RecordParticipantPrepares, - TransactionCommandInput, TransactionFailure, account_target, coordinator_target, data_target, + CommitPreparedTransaction, CommitPreparedTransactionInput, CommitPreparedTransactionOutcome, + CoordinatorDecision, CoordinatorParticipant, CoordinatorParticipantTarget, + CoordinatorPhaseOutcome, CoordinatorPrepareReceipt, CrossCellTransactionStatus, + DecideCrossCellTransaction, DecideCrossCellTransactionInput, DecideCrossCellTransactionOutcome, + Json, ParticipantTransactionState, PrepareAccountTransaction, PrepareAccountTransactionBounded, + PrepareAccountTransactionInput, PreparePartitionTransaction, + PreparePartitionTransactionBounded, PreparePartitionTransactionInput, + PrepareTransactionOutcome, ReadCoordinatorParticipantInput, ReadCoordinatorResume, + ReadCrossCellTransaction, ReadCrossCellTransactionInput, ReadTransactionInput, + ReadUnresolvedCoordinatorParticipants, TransactionCommandInput, TransactionFailure, + TransactionOperation, UnresolvedCoordinatorParticipant, account_target, coordinator_target, + data_target, }; impl CellStorage { @@ -50,53 +54,138 @@ impl CellStorage { transaction_id, routing_key: routing_key.to_vec(), }; - if self.transaction_status(&coordinator, &read).await?.decision - != CoordinatorDecision::Begin - { - return self.finish_transaction(&coordinator, &read).await; - } - let participants = self + let snapshot = self .client - .query::(&coordinator, None, Json(read.clone())) + .query::(&coordinator, None, Json(read.clone())) .await .map_err(cell_error)? .output .0; - // Fetch immutable participant payloads concurrently. Each participant - // has its own durable cell, so these reads do not need to be serialized. - let payloads = stream::iter( + let status = snapshot + .status + .ok_or_else(|| StorageError::Internal("coordinator transaction is missing".into()))?; + if status.decision != CoordinatorDecision::Begin { + return self.finish_transaction(&coordinator, &read).await; + } + // The Cell query returns status and bounded immutable operations from + // the same observation. Large payloads retain chunked retrieval. + let payloads = if let Some(participants) = snapshot.participants { participants .into_iter() - .filter(|participant| { - // Durable prepare evidence survives driver and owner replacement. - // Keep these participants in recovery's list until resolution, but - // do not re-upload their payloads or publish another prepare receipt. - !participant.prepared - }) - .map(|participant| { - let participant_coordinator = coordinator.clone(); - let input = ReadCoordinatorParticipantInput { - account_id: read.account_id.clone(), - transaction_id, - routing_key: read.routing_key.clone(), - position: participant.position, - chunk: 0, - }; - async move { - let payload = self - .coordinator_participant(&participant_coordinator, input) - .await?; - Ok::<_, StorageError>((participant.position, payload)) - } - }), - ) - .buffer_unordered(8) - .collect::>() - .await; + .map(|entry| Ok((entry.position, Some(entry.participant)))) + .collect::>>() + } else { + let participants = self + .client + .query::( + &coordinator, + None, + Json(read.clone()), + ) + .await + .map_err(cell_error)? + .output + .0; + // Fetch immutable coordinator chunks in a bounded window. The + // participant targets remain the durable transaction authority. + stream::iter( + participants + .into_iter() + .filter(|participant| { + // Durable prepare evidence survives driver and owner replacement. + // Keep these participants in recovery's list until resolution, but + // do not re-upload their payloads or publish another prepare receipt. + !participant.prepared + }) + .map(|participant| { + let participant_coordinator = coordinator.clone(); + let input = ReadCoordinatorParticipantInput { + account_id: read.account_id.clone(), + transaction_id, + routing_key: read.routing_key.clone(), + position: participant.position, + chunk: 0, + }; + async move { + let payload = self + .coordinator_participant(&participant_coordinator, input) + .await?; + Ok::<_, StorageError>((participant.position, payload)) + } + }), + ) + .buffer_unordered(8) + .collect::>() + .await + }; + self.prepare_and_commit_transaction(&coordinator, &read, payloads, None) + .await + } + + /// Drive only the exact bounded payload of an acknowledged fresh BEGIN. + /// Retries and recovery enter through the durable coordinator snapshot path. + pub(super) async fn drive_acknowledged_transaction( + &self, + coordinator: &CellTarget, + read: &ReadCrossCellTransactionInput, + participants: Vec, + ) -> Result { + // Only an acknowledged fresh BEGIN establishes this complete immutable + // participant set. Reads retain the durable read-image cleanup path. + let write_participants = if participants.iter().all(|participant| { + participant + .operations + .iter() + .all(|operation| !matches!(operation.operation, TransactionOperation::Read(_))) + }) { + Some( + participants + .iter() + .enumerate() + .map(|(position, participant)| { + Ok(UnresolvedCoordinatorParticipant { + position: u8::try_from(position).map_err(|_| { + StorageError::Internal( + "invalid acknowledged participant position".into(), + ) + })?, + target: participant.target.clone(), + prepared: true, + release_read_result: false, + }) + }) + .collect::, StorageError>>()?, + ) + } else { + None + }; + let payloads = participants + .into_iter() + .enumerate() + .map(|(position, participant)| { + let position = u8::try_from(position).map_err(|_| { + StorageError::Internal("invalid acknowledged participant position".into()) + })?; + Ok((position, Some(participant))) + }) + .collect(); + self.prepare_and_commit_transaction(coordinator, read, payloads, write_participants) + .await + } + + async fn prepare_and_commit_transaction( + &self, + coordinator: &CellTarget, + read: &ReadCrossCellTransactionInput, + payloads: Vec), StorageError>>, + write_participants: Option>, + ) -> Result { + let account_id = read.account_id.as_str(); + let transaction_id = read.transaction_id; let mut attempts = Vec::new(); for payload in payloads { let (position, Some(payload)) = payload? else { - return self.finish_transaction(&coordinator, &read).await; + return self.finish_transaction(coordinator, read).await; }; let operations = payload .operations @@ -175,8 +264,8 @@ impl CellStorage { // decides, and all resolutions finish before cancellation returns. return self .decide_transaction( - &coordinator, - &read, + coordinator, + read, CoordinatorDecision::Abort { index: Some(operation.index), reason: Some(TransactionFailure::Throttled), @@ -199,13 +288,18 @@ impl CellStorage { PrepareTransactionOutcome::Committed | PrepareTransactionOutcome::Aborted => { // A concurrent driver/recovery owner may have finished. Only // the coordinator can decide which terminal outcome to return. - return self.finish_transaction(&coordinator, &read).await; + return self.finish_transaction(coordinator, read).await; } PrepareTransactionOutcome::Mismatch => { return Err(StorageError::Internal( "participant transaction identity mismatch".into(), )); } + PrepareTransactionOutcome::WideRequired => { + return Err(StorageError::Internal( + "prepare reply fallback was not resolved".into(), + )); + } }; if let Some((index, reason)) = rejection { let operation = payload.operations.get(index).ok_or_else(|| { @@ -215,59 +309,85 @@ impl CellStorage { index: Some(operation.index), reason: Some(reason), }; - return self.decide_transaction(&coordinator, &read, decision).await; + return self.decide_transaction(coordinator, read, decision).await; } evidence.push((position, target, receipt)); } - // Participant prepares already ran concurrently. Record their receipts - // in one coordinator command so the coordinator publishes one durable - // evidence update instead of one round trip per participant. - let phases: Vec<_> = evidence + // All participant receipts are already durable. Record their evidence + // and COMMIT in one coordinator command, retaining the existing check + // that every participant has a prepare receipt before deciding. + let prepares = evidence .into_iter() - .map(|(position, target, receipt)| CoordinatorPhaseInput { - account_id: read.account_id.clone(), - transaction_id, - routing_key: read.routing_key.clone(), + .map(|(position, target, receipt)| CoordinatorPrepareReceipt { position, participant_cell: *target.cell_id().as_bytes(), sequence: receipt.commit_sequence, }) .collect(); - if phases.is_empty() { - return self - .decide_transaction(&coordinator, &read, CoordinatorDecision::Commit) - .await; - } - let recorded = self + let result = self .client - .command::(&coordinator, mutation_identity()?, Json(phases)) + .command::( + coordinator, + mutation_identity()?, + Json(CommitPreparedTransactionInput { + transaction: read.clone(), + prepares, + }), + ) .await; - match recorded { - Ok(result) - if result.output.0.iter().all(|outcome| { - matches!( - outcome, - CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay + let outcome = match result { + Ok(result) => { + if result.output.0 + == CommitPreparedTransactionOutcome::Decision( + DecideCrossCellTransactionOutcome::Decided(CoordinatorDecision::Commit), ) - }) => {} - Ok(result) - if result - .output - .0 - .contains(&CoordinatorPhaseOutcome::WrongDecision) => - { - return self.finish_transaction(&coordinator, &read).await; + && let Some(participants) = write_participants + { + let participant_count = u8::try_from(participants.len()).map_err(|_| { + StorageError::Internal("invalid acknowledged participant count".into()) + })?; + // The accepted COMMIT records every prepare receipt. The + // exact fresh BEGIN set and acknowledged resolution records + // then prove completion, without rediscovering that set or + // polling coordinator status. Any uncertainty below returns + // an error; retries use the normal durable snapshot path. + self.finish_transaction_participants(coordinator, read, true, participants) + .await?; + return Ok(CrossCellTransactionStatus { + decision: CoordinatorDecision::Commit, + participant_count, + prepared_count: participant_count, + resolved_count: participant_count, + unreleased_read_results: 0, + }); + } + result.output.0 } - Ok(_) | Err(InvocationError::Rejected(_)) => { - return Err(StorageError::Internal( - "coordinator refused prepare evidence".into(), - )); + Err(InvocationError::Rejected(result)) => result.output.0, + // An absent reply cannot distinguish a durable decision from an + // unfinished command. Read authoritative state before resolving. + Err(InvocationError::Pending(_)) => { + return self.finish_transaction(coordinator, read).await; } Err(error) => return Err(cell_error(error)), + }; + match outcome { + CommitPreparedTransactionOutcome::Decision( + DecideCrossCellTransactionOutcome::Decided(_) + | DecideCrossCellTransactionOutcome::DecisionConflict, + ) => self.finish_transaction(coordinator, read).await, + CommitPreparedTransactionOutcome::EvidenceRejected(outcomes) + if outcomes.contains(&CoordinatorPhaseOutcome::WrongDecision) => + { + // A competing driver may have committed or aborted while the + // prepares were in flight. Its durable decision is authoritative. + self.finish_transaction(coordinator, read).await + } + _ => Err(StorageError::Internal( + "coordinator refused prepared transaction commit".into(), + )), } - self.decide_transaction(&coordinator, &read, CoordinatorDecision::Commit) - .await } async fn transaction_status( @@ -358,21 +478,31 @@ impl CellStorage { // the participant Cell. Avoid a separate state query on the normal // first-attempt path; only an ambiguous reply needs a follow-up read. let identity = mutation_identity()?; - let inline = match &input { + let input_bytes = match &input { ParticipantPrepare::Account(input) => serde_json::to_vec(input), ParticipantPrepare::Data(input) => serde_json::to_vec(input), } .map_err(|error| StorageError::Internal(error.to_string()))? - .len() - <= crate::transaction_transport::INLINE_BYTES; + .len(); + let inline = input_bytes <= crate::transaction_transport::INLINE_BYTES; + let bounded = input_bytes <= crate::participant::SMALL_PREPARE_BYTES; + if let ParticipantPrepare::Data(ref input) = input + && bounded + { + return self + .prepare_batcher + .submit(target.clone(), input.clone()) + .await; + } let result = match input { ParticipantPrepare::Account(input) => { if inline { - self.client - .command::( + self + .prepare_inline::( target, identity, - Json(TransactionCommandInput::Inline(input)), + input, + bounded, ) .await } else { @@ -390,11 +520,12 @@ impl CellStorage { } ParticipantPrepare::Data(input) => { if inline { - self.client - .command::( + self + .prepare_inline::( target, identity, - Json(TransactionCommandInput::Inline(input)), + input, + bounded, ) .await } else { diff --git a/src/backend/transaction_read.rs b/src/backend/transaction_read.rs index 0e583da..f4f1fa2 100644 --- a/src/backend/transaction_read.rs +++ b/src/backend/transaction_read.rs @@ -9,13 +9,14 @@ use std::time::Duration; use super::{CellStorage, cell_error, mutation_identity}; use crate::{ - BeginReadResultRelease, CoordinatorDecision, CoordinatorParticipantTarget, - CoordinatorPhaseOutcome, GetItemInput, Json, ReadAccountTransactionResult, - ReadCoordinatorParticipantInput, ReadPartitionTransactionResult, ReadResultRelease, - ReadTransactionInput, ReadTransactionResultInput, RecordReadResultReleases, - RecordReadResultReleasesInput, ReleaseAccountTransactionReads, - ReleasePartitionTransactionReads, TransactionFailure, TransactionOperation, - TransactionReadResult, account_target, coordinator_target, data_target, + BeginReadResultRelease, BoundedTransactionReadResult, CoordinatorDecision, + CoordinatorParticipantTarget, CoordinatorPhaseOutcome, GetItemInput, Json, + ReadAccountTransactionResult, ReadAccountTransactionResultBounded, + ReadCoordinatorParticipantInput, ReadPartitionTransactionResult, + ReadPartitionTransactionResultBounded, ReadResultRelease, ReadTransactionInput, + ReadTransactionResultInput, RecordReadResultReleases, RecordReadResultReleasesInput, + ReleaseAccountTransactionReads, ReleasePartitionTransactionReads, TransactionFailure, + TransactionOperation, TransactionReadResult, account_target, coordinator_target, data_target, }; #[derive(Clone)] @@ -25,6 +26,49 @@ enum ReadTarget { } impl CellStorage { + async fn saved_transaction_image( + &self, + target: &ReadTarget, + input: Json, + ) -> Result { + let result = match target { + ReadTarget::Account(target) => { + self.client + .query::(target, None, input.clone()) + .await + } + ReadTarget::Data(target) => { + self.client + .query::(target, None, input.clone()) + .await + } + } + .map_err(cell_error)? + .output; + match result { + BoundedTransactionReadResult::Unavailable => Ok(TransactionReadResult::Unavailable), + BoundedTransactionReadResult::Item(item) => Ok(TransactionReadResult::Item(item)), + BoundedTransactionReadResult::WideRequired => { + // Only the explicit successful small reply permits a wide query. + // Keep the same coordinator identity and saved-image position. + let result = match target { + ReadTarget::Account(target) => { + self.client + .query::(target, None, input) + .await + } + ReadTarget::Data(target) => { + self.client + .query::(target, None, input) + .await + } + } + .map_err(cell_error)?; + Ok(result.output.0) + } + } + } + pub(super) async fn transaction_read( &self, account_id: &str, @@ -55,24 +99,48 @@ impl CellStorage { let identity = admitted.identity; let coordinator = coordinator_target(account_id, &identity.routing_key) .map_err(|error| StorageError::Internal(error.to_string()))?; - let participant_inputs = - (0..admitted.participant_count).map(|position| ReadCoordinatorParticipantInput { - account_id: account_id.into(), - transaction_id: identity.transaction_id, - routing_key: identity.routing_key.clone(), - position, - chunk: 0, - }); - let participant_results = stream::iter(participant_inputs.map(|input| async { - let position = input.position; - ( - position, - self.coordinator_participant(&coordinator, input).await, - ) - })) - .buffer_unordered(8) - .collect::>() - .await; + let participant_results = if let Some(participants) = + admitted.acknowledged_read_participants + { + if participants.len() != usize::from(admitted.participant_count) { + return Err(StorageError::Internal( + "acknowledged read participant count differs".into(), + )); + } + // The acknowledged fresh BEGIN fixes this exact ordering. Reuse + // only routing/operation metadata; images still come from durable + // participant snapshots, never from live items or current routes. + participants + .into_iter() + .enumerate() + .map(|(position, participant)| { + u8::try_from(position) + .map(|position| (position, Ok(Some(participant)))) + .map_err(|_| { + StorageError::Internal("invalid read participant position".into()) + }) + }) + .collect::, _>>()? + } else { + let participant_inputs = + (0..admitted.participant_count).map(|position| ReadCoordinatorParticipantInput { + account_id: account_id.into(), + transaction_id: identity.transaction_id, + routing_key: identity.routing_key.clone(), + position, + chunk: 0, + }); + stream::iter(participant_inputs.map(|input| async { + let position = input.position; + ( + position, + self.coordinator_participant(&coordinator, input).await, + ) + })) + .buffer_unordered(8) + .collect::>() + .await + }; let mut image_reads: HashMap<[u8; 32], Vec<_>> = HashMap::new(); let mut participant_targets = Vec::with_capacity(participant_results.len()); for (participant_position, result) in participant_results { @@ -116,27 +184,13 @@ impl CellStorage { )); } } - // One image query reserves up to the Cell wire result ceiling. Fetch - // each participant's images serially to stay inside its 16 MiB mailbox, - // while independent participant Cells can still make progress together. + // Small saved images reserve a bounded reply instead of the wide item + // ceiling. Keep each participant's reads serial for large-image fallback; + // independent participants can still make progress together. let images = stream::iter(image_reads.into_values().map(|reads| async move { let mut group = Vec::with_capacity(reads.len()); for (index, target, input) in reads { - let image = match target { - ReadTarget::Account(target) => { - self.client - .query::(&target, None, input) - .await - } - ReadTarget::Data(target) => { - self.client - .query::(&target, None, input) - .await - } - } - .map_err(cell_error)? - .output - .0; + let image = self.saved_transaction_image(&target, input).await?; let TransactionReadResult::Item(image) = image else { return Err(StorageError::Internal( "committed read image is missing".into(), diff --git a/src/backend/transaction_transport.rs b/src/backend/transaction_transport.rs index 503f1bf..78f8edf 100644 --- a/src/backend/transaction_transport.rs +++ b/src/backend/transaction_transport.rs @@ -1,12 +1,18 @@ //! Bounded phase inputs and immutable coordinator payload retrieval. -use cellule_runtime::{MutationIdentity, client::InvocationError, identity::CellTarget}; +use cellule_runtime::{ + MutationIdentity, + client::{Committed, InvocationError}, + identity::{CellTarget, RequestId}, + registry::Command, +}; use extenddb_storage::error::StorageError; use super::{CellStorage, cell_error, mutation_identity}; use crate::{ - CoordinatorParticipant, Json, MultipartTransactionCommand, ReadCoordinatorParticipant, - ReadCoordinatorParticipantInput, TransactionPayloadRef, UploadTransactionPayload, + CoordinatorParticipant, Json, MultipartTransactionCommand, PrepareTransactionOutcome, + ReadCoordinatorParticipant, ReadCoordinatorParticipantInput, TransactionCommandInput, + TransactionPayloadRef, UploadTransactionPayload, }; // Preserve proven capacity refusal until the transaction driver can decide @@ -50,6 +56,54 @@ impl From for StorageError { } impl CellStorage { + pub(super) async fn prepare_inline( + &self, + target: &CellTarget, + identity: MutationIdentity, + input: C::Payload, + bounded: bool, + ) -> Result< + Committed>, + InvocationError>, + > + where + C: MultipartTransactionCommand>, + C::Payload: Clone + Send, + B: Command, Output = Json>, + { + let identity = if bounded { + let result = self + .client + .command::(target, identity, Json(input.clone())) + .await; + if !matches!( + &result, + Err(InvocationError::Rejected(committed)) + if committed.output.0 == PrepareTransactionOutcome::WideRequired + ) { + // Pending, invalid results and capacity errors retain their + // original meaning. Only a durable rejected receipt permits + // fallback; never infer rollback from a missing participant. + return result; + } + MutationIdentity { + // A different opcode has a different mutation digest. Retain + // the phase deadline but never reuse its request identity. + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + ..identity + } + } else { + identity + }; + self.client + .command::( + target, + identity, + Json(TransactionCommandInput::Inline(input)), + ) + .await + } + pub(super) async fn upload_transaction( &self, target: &CellTarget, diff --git a/src/backend/update_batch.rs b/src/backend/update_batch.rs new file mode 100644 index 0000000..81a826f --- /dev/null +++ b/src/backend/update_batch.rs @@ -0,0 +1,241 @@ +//! Bounded coalescing for independent updates with returned item images. + +use std::collections::{HashMap, VecDeque}; +use std::sync::{ + Arc, Weak, + atomic::{AtomicBool, Ordering}, +}; +use std::time::Duration; + +use cellule_runtime::client::InvocationError; +use cellule_runtime::identity::CellTarget; +use extenddb_storage::error::StorageError; +use tokio::sync::{Mutex, OwnedSemaphorePermit, Semaphore, oneshot}; + +use super::{cell_error, mutation_identity}; +use crate::partition::{BATCH_INPUT_BYTES, MAX_UPDATES}; +use crate::{ + BatchedPartitionUpdate, Json, PartitionUpdateBatch, PartitionUpdateBatchOutcome, + PartitionUpdateIndividual, PartitionUpdateOutcome, +}; + +const WINDOW: Duration = Duration::from_millis(2); +const MAX_PENDING: usize = 64; + +struct Pending { + update: BatchedPartitionUpdate, + key: Vec, + bytes: usize, + reply: oneshot::Sender>, + _permit: OwnedSemaphorePermit, +} + +struct Slot { + target: CellTarget, + queued: Mutex>, + permits: Arc, + scheduled: AtomicBool, +} + +pub(crate) struct UpdateBatcher { + client: cellule_runtime::CellClient, + slots: Mutex>>, +} + +impl super::CellStorage { + pub(crate) async fn submit_update( + &self, + target: CellTarget, + key: Vec, + update: BatchedPartitionUpdate, + ) -> Result { + self.update_batcher.submit(target, key, update).await + } +} + +impl UpdateBatcher { + pub(crate) fn new(client: cellule_runtime::CellClient) -> Self { + Self { + client, + slots: Mutex::new(HashMap::new()), + } + } + + async fn submit( + self: &Arc, + target: CellTarget, + key: Vec, + update: BatchedPartitionUpdate, + ) -> Result { + let bytes = serde_json::to_vec(&update) + .map_err(|error| StorageError::Internal(error.to_string()))? + .len() + + 1; + if bytes + 6 > BATCH_INPUT_BYTES as usize { + return self.individual(&target, update).await; + } + let slot = { + let mut slots = self.slots.lock().await; + let id = *target.cell_id().as_bytes(); + if let Some(slot) = slots.get(&id).and_then(Weak::upgrade) { + slot + } else { + slots.retain(|_, slot| slot.strong_count() > 0); + let slot = Arc::new(Slot { + target, + queued: Mutex::new(VecDeque::new()), + permits: Arc::new(Semaphore::new(MAX_PENDING)), + scheduled: AtomicBool::new(false), + }); + slots.insert(id, Arc::downgrade(&slot)); + slot + } + }; + // Backpressure precedes enqueueing. Permits cover the selected batch + // until its durable result is delivered as well as queued operations. + let permit = slot + .permits + .clone() + .acquire_owned() + .await + .map_err(|_| StorageError::Transient("update batch queue closed".into()))?; + let (reply, result) = oneshot::channel(); + let schedule = { + let mut queued = slot.queued.lock().await; + queued.push_back(Pending { + update, + key, + bytes, + reply, + _permit: permit, + }); + !slot.scheduled.swap(true, Ordering::AcqRel) + }; + if schedule { + let batcher = self.clone(); + tokio::spawn(async move { + batcher.flush(slot).await; + }); + } + result + .await + .map_err(|_| StorageError::Transient("update batch worker stopped".into()))? + } + + async fn flush(self: Arc, slot: Arc) { + loop { + if slot.queued.lock().await.len() < MAX_UPDATES { + tokio::time::sleep(WINDOW).await; + } + let pending = { + let mut queue = slot.queued.lock().await; + let Some(first) = queue.front() else { + slot.scheduled.store(false, Ordering::Release); + return; + }; + let table_id = first.update.input.table_id.clone(); + let epoch = first.update.input.epoch; + let mut selected: Vec = Vec::new(); + let mut deferred = VecDeque::new(); + let mut bytes = 6; + let count = queue.len(); + while selected.len() < MAX_UPDATES && selected.len() + deferred.len() < count { + let Some(candidate) = queue.front() else { + break; + }; + if candidate.update.input.table_id != table_id + || candidate.update.input.epoch != epoch + || bytes + candidate.bytes > BATCH_INPUT_BYTES as usize + { + break; + } + let Some(candidate) = queue.pop_front() else { + break; + }; + if selected.iter().any(|prior| prior.key == candidate.key) { + deferred.push_back(candidate); + } else { + bytes += candidate.bytes; + selected.push(candidate); + } + } + while let Some(candidate) = deferred.pop_back() { + queue.push_front(candidate); + } + selected + }; + let identity = match mutation_identity() { + Ok(identity) => identity, + Err(error) => { + for pending in pending { + let _ = pending.reply.send(Err(error.clone())); + } + continue; + } + }; + let updates = pending + .iter() + .map(|pending| pending.update.clone()) + .collect(); + let result = self + .client + .command::(&slot.target, identity, Json(updates)) + .await; + match result { + Ok(committed) => match committed.output { + PartitionUpdateBatchOutcome::Results(results) + if results.len() == pending.len() => + { + for (pending, result) in pending.into_iter().zip(results) { + let _ = pending.reply.send(Ok(result)); + } + } + _ => { + for pending in pending { + let _ = pending.reply.send(Err(StorageError::Internal( + "invalid committed update batch reply".into(), + ))); + } + } + }, + Err(InvocationError::Rejected(committed)) + if committed.output == PartitionUpdateBatchOutcome::IndividualRequired => + { + // A confirmed rejected receipt proves the complete batch + // rolled back. Never retry after an ambiguous invocation. + for pending in pending { + let result = self.individual(&slot.target, pending.update).await; + let _ = pending.reply.send(result); + } + } + Err(error) => { + let error = cell_error(error); + for pending in pending { + let _ = pending.reply.send(Err(error.clone())); + } + } + } + } + } + + async fn individual( + &self, + target: &CellTarget, + update: BatchedPartitionUpdate, + ) -> Result { + let output = self + .client + .command::(target, mutation_identity()?, Json(vec![update])) + .await + .map_err(cell_error)? + .output; + match output { + PartitionUpdateBatchOutcome::Results(mut results) if results.len() == 1 => results + .pop() + .ok_or_else(|| StorageError::Internal("missing individual update reply".into())), + _ => Err(StorageError::Internal( + "invalid individual update reply".into(), + )), + } + } +} diff --git a/src/bin/beyonddb.rs b/src/bin/beyonddb.rs index fc03ff3..ac71f22 100644 --- a/src/bin/beyonddb.rs +++ b/src/bin/beyonddb.rs @@ -1,14 +1,29 @@ //! Run ExtendDB's signed DynamoDB endpoint over a leased BeyondDB Cell node. -use std::{error::Error, io, io::Read, net::SocketAddr, path::PathBuf, sync::Arc, time::Duration}; +use std::{ + error::Error, + io, + io::Read, + net::SocketAddr, + path::PathBuf, + sync::{ + Arc, + atomic::{AtomicBool, Ordering}, + }, + time::Duration, +}; use beyonddb::{ APPLICATION_ID, Beyonddb, BeyonddbPeers, CellAuthorizationStore, CellCredentialStore, - CellInitialPartitionProvisioner, CellStorage, NodeLeasePublisher, PeerNodeLogTransport, - build_http_state_with_cache, measured_node_capacity, shutdown_serving_node, + CellInitialPartitionProvisioner, CellStorage, NodeLeasePublisher, PeerNodeDurabilityProvider, + PeerNodeLogTransport, RuntimeMetrics, build_http_state_with_cache, measured_node_capacity, + shutdown_serving_node, }; use cellule_app::CellApplication; -use cellule_host::{CellNode, CellNodeBuilder, CellNodeTaskGroup, FOLLOWER_STORE_COMPONENT}; +use cellule_host::{ + CellNode, CellNodeBuilder, CellNodeTaskGroup, FOLLOWER_STORE_COMPONENT, + NodeDurabilitySupervisorConfig, +}; use cellule_peer_http::{LoadedPeerTls, PeerTlsIdentity}; use cellule_runtime::{ NodeLeaseGuard, SqlWorkerPool, @@ -21,7 +36,11 @@ use cellule_runtime::{ }, registry::BuildDescriptor, }; -use cellule_store::{Store, provider_store::build_url_object_store}; +use cellule_store::{ + Store, + identity::StorageProviderKind, + provider_store::{build_static_env_store, build_url_object_store}, +}; use extenddb_auth::StoredCredential; use extenddb_server::ServerTlsConfig; use serde::Deserialize; @@ -42,8 +61,14 @@ struct Config { /// Optional persistent follower-lane budget. Does not enable fleet proofs. #[serde(default)] follower_store_bytes: Option, + /// Experimental follower fsync proofs with fenced-owner recovery. + #[serde(default)] + follower_durability_enabled: bool, #[serde(default = "default_node_retained_bytes")] node_retained_bytes: usize, + /// Optional bounded runtime observations, atomically refreshed once per second. + #[serde(default)] + runtime_metrics_file: Option, #[serde(default = "default_max_active_cells")] max_active_cells: usize, /// Optional SQL worker override. The runtime caps this at sixteen workers. @@ -104,6 +129,7 @@ const fn default_max_active_cells() -> usize { } const MAX_SQL_WORKERS: usize = 16; +const REQUIRED_FOLLOWER_BYTES: u64 = 64 * 1024 * 1024; #[tokio::main] async fn main() -> ServerResult<()> { @@ -154,6 +180,16 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S ) .into()); } + if config.follower_durability_enabled + && config + .follower_store_bytes + .is_none_or(|bytes| bytes < REQUIRED_FOLLOWER_BYTES) + { + return Err(invalid( + "follower durability requires a persistent follower-store budget of at least 64 MiB", + ) + .into()); + } if !["s3://", "gs://", "az://"] .iter() .any(|scheme| config.storage_url.starts_with(scheme)) @@ -209,12 +245,33 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S })?); let release = application.registry().release_digest(); let image = application.descriptor_digest(); - let url_store = build_url_object_store(&config.storage_url)?; - let layout = CellStorageLayout::new( - Store::new(url_store.store_arc()), - url_store.prefix().clone(), - *APPLICATION_ID.as_bytes(), - ); + let (store, prefix) = if config.storage_url.starts_with("s3://") { + let url = reqwest::Url::parse(&config.storage_url)?; + let bucket = url + .host_str() + .ok_or_else(|| invalid("S3 bucket is missing"))?; + // Recovery pins immutable overlays with conditional copies. The provider + // builder configures multipart copy-if-absent and transport admission. + ( + build_static_env_store(bucket, StorageProviderKind::S3)?, + object_store::path::Path::from_url_path(url.path())?, + ) + } else { + let url_store = build_url_object_store(&config.storage_url)?; + ( + Store::new(url_store.store_arc()), + url_store.prefix().clone(), + ) + }; + let metrics = config + .runtime_metrics_file + .as_ref() + .map(|_| Arc::new(RuntimeMetrics::default())); + let store = match &metrics { + Some(metrics) => store.with_storage_observer(metrics.clone()), + None => store, + }; + let layout = CellStorageLayout::new(store, prefix, *APPLICATION_ID.as_bytes()); let directory = NodeDirectory::new(layout.clone(), tls.fleet(), image, release); let peer_listener = TcpListener::bind(config.peer_bind).await?; let public_listener = TcpListener::bind(config.public_bind).await?; @@ -242,6 +299,46 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S let node = builder.build()?; let node_shutdown = CancellationToken::new(); let tasks = node.install_task_group(CancellationToken::new(), node_shutdown.clone())?; + if let Some((path, metrics)) = config.runtime_metrics_file.clone().zip(metrics.clone()) { + node.install_telemetry(metrics.clone())?; + let metrics_runtime = node.runtime(); + let cancellation = tasks.cancellation_token(); + tasks.spawn(async move { + let mut tick = tokio::time::interval(Duration::from_secs(1)); + tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + let mut pending = path.as_os_str().to_os_string(); + pending.push(".pending"); + let pending = PathBuf::from(pending); + loop { + tokio::select! { + () = cancellation.cancelled() => return Ok::<(), std::io::Error>(()), + _ = tick.tick() => {} + } + let mut snapshot = metrics.snapshot(); + let stats = metrics_runtime.stats(); + snapshot["node_resources"] = serde_json::json!({ + "active_cells": stats.active_cells(), + "active_cell_capacity": stats.active_cell_capacity(), + "retained_bytes": stats.retained_bytes(), + "retained_capacity_bytes": stats.retained_capacity_bytes(), + "worker_jobs": stats.worker_jobs(), + "worker_job_capacity": stats.worker_job_capacity(), + "unpublished_node_log_bytes": stats.unpublished_node_log_bytes(), + }); + let encoded = serde_json::to_vec_pretty(&snapshot); + let result = match encoded { + Ok(encoded) => match tokio::fs::write(&pending, encoded).await { + Ok(()) => tokio::fs::rename(&pending, &path).await, + Err(error) => Err(error), + }, + Err(error) => Err(std::io::Error::other(error)), + }; + if let Err(error) = result { + tracing::warn!(%error, "runtime metrics snapshot failed"); + } + } + })?; + } let node_id = NodeId::from_bytes(*config.node_id.as_bytes()); let endpoint = config.peer_endpoint.clone(); let signer = tls.signing_key().clone(); @@ -250,6 +347,12 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S let capacity_runtime = node.runtime(); let capacity_dir = config.data_dir.clone(); let recovery_capable = config.follower_store_bytes.is_some(); + let follower_store = node.owned_component::(FOLLOWER_STORE_COMPONENT); + let follower_ready = Arc::new(AtomicBool::new(false)); + let recruitment_ready = Arc::new(AtomicBool::new(false)); + let advertised_store = follower_store.clone(); + let advertised_ready = follower_ready.clone(); + let follower_durability_enabled = config.follower_durability_enabled; let modules = application.registry().module_digests(); let published = NodeLeasePublisher::new(directory.clone(), move |now, expires| { let (capacity, placement) = if capacity_runtime.is_shutting_down() { @@ -257,12 +360,17 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S } else { match measured_node_capacity(&capacity_dir, capacity_runtime.stats()) { Ok((mut capacity, placement)) => { - // A persistent follower store permits fenced recovery claims. - // Zero advertised follower bytes still prevents recruitment - // until follower-backed serving has a tested recovery path. if recovery_capable { capacity.log_protocol = NODE_LOG_PROTOCOL_VERSION; } + // Only a listening authenticated receiver may be recruited. + if follower_durability_enabled + && advertised_ready.load(Ordering::Acquire) + && let Some(store) = advertised_store.as_ref() + { + capacity.follower_free_bytes = + store.available_bytes().min(capacity.free_disk_bytes); + } (capacity, Some(placement)) } Err(error) => { @@ -299,6 +407,42 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S .await?; let follower_guard = published.guard(); node.install_node_lease_for_startup(follower_guard.clone())?; + if config.follower_durability_enabled { + let store = + follower_store.ok_or_else(|| invalid("persistent follower store is unavailable"))?; + let mut transport = PeerNodeLogTransport::new( + directory.clone(), + tls.client_identity(), + session, + node_id, + follower_guard.clone(), + ) + .with_local_follower_store(store); + if let Some(metrics) = &metrics { + transport = transport.with_runtime_metrics(metrics.clone()); + } + let provider = PeerNodeDurabilityProvider::new( + published.log_authority(), + transport, + session, + node_id, + follower_guard.clone(), + node.runtime().telemetry_handle(), + )? + .with_recruitment_gate(recruitment_ready.clone()); + node.install_node_durability_provider( + Arc::new(provider), + NodeDurabilitySupervisorConfig::new( + APPLICATION_ID, + Limits::default(), + REQUIRED_FOLLOWER_BYTES, + 1024, + Duration::from_secs(1), + Duration::from_secs(30), + 65_536, + )?, + )?; + } // Publication during drain still needs the node lease. The host cancels // lease maintenance only after the runtime and its durable log close. tasks.spawn_lease_maintenance(async move { published.run(&node_shutdown).await })?; @@ -320,6 +464,9 @@ async fn serve(config: Config, bootstrap_secret: Option>) -> S public_tls, encryption_key, bootstrap_secret, + follower_ready, + recruitment_ready, + metrics, ) .await; let shutdown = shutdown_serving_node(&node, &directory, session).await; @@ -356,21 +503,29 @@ async fn serve_ready( public_tls: Option, encryption_key: [u8; 32], bootstrap_secret: Option>, + follower_ready: Arc, + recruitment_ready: Arc, + metrics: Option>, ) -> ServerResult<()> { let mut peers = BeyonddbPeers::new(node, layout.clone(), directory.clone(), session, &tls)?; + if let Some(metrics) = &metrics { + peers = peers.with_follower_metrics(metrics.clone()); + } let mut recovery_transport = None; if let Some(store) = node.owned_component::(FOLLOWER_STORE_COMPONENT) { let node_id = NodeId::from_bytes(*config.node_id.as_bytes()); - recovery_transport = Some(Arc::new( - PeerNodeLogTransport::new( - directory.clone(), - tls.client_identity(), - session, - node_id, - follower_guard.clone(), - ) - .with_local_follower_store(store.clone()), - )); + let mut transport = PeerNodeLogTransport::new( + directory.clone(), + tls.client_identity(), + session, + node_id, + follower_guard.clone(), + ) + .with_local_follower_store(store.clone()); + if let Some(metrics) = &metrics { + transport = transport.with_runtime_metrics(metrics.clone()); + } + recovery_transport = Some(Arc::new(transport)); peers = peers.with_follower_store(node_id, store, follower_guard); } let peers = Arc::new(peers); @@ -438,7 +593,7 @@ async fn serve_ready( state.tls_enabled = public_tls.is_some(); let peer_cancel = CancellationToken::new(); let peer_shutdown = peer_cancel.clone(); - let peer_router = peers.router(provisioner.clone()); + let peer_router = peers.router_with_cache(provisioner.clone(), config.auth_cache_enabled); // Other recovering nodes may need these local participants. Start private // routing before resolving decisions, while public requests remain gated. let mut peer_server = tokio::spawn(async move { @@ -449,6 +604,7 @@ async fn serve_ready( .with_graceful_shutdown(async move { peer_shutdown.cancelled().await }) .await }); + follower_ready.store(true, Ordering::Release); let recovery: ServerResult<()> = async { let storage = CellStorage::new(client.clone(), config.region.clone()); for account_id in &config.owned_accounts { @@ -538,6 +694,7 @@ async fn serve_ready( peer_server.await??; return Err(error); } + recruitment_ready.store(true, Ordering::Release); let mut public_server = tokio::spawn(extenddb_server::start_server( public_listener, state, diff --git a/src/credentials.rs b/src/credentials.rs index eed1b7d..8c7f9f4 100644 --- a/src/credentials.rs +++ b/src/credentials.rs @@ -3,6 +3,7 @@ use std::{ collections::HashSet, sync::{OnceLock, RwLock}, + time::Duration, }; use aes_gcm::aead::{Aead, Payload}; @@ -31,6 +32,7 @@ pub(crate) const NAMESPACE: NamespaceId = NamespaceId::from_bytes([0x44; 16]); const TENANT: TenantId = TenantId::from_bytes([0x44; 16]); const SHARDS: u32 = 256; const SCHEMA: &str = include_str!("credential_schema.sql"); +const QUERY_TIMEOUT: Duration = Duration::from_secs(30); static NAMESPACES: [NamespaceDescriptor; 1] = [NamespaceDescriptor { id: NAMESPACE, @@ -408,11 +410,35 @@ impl CellCredentialStore { } self.remember_cell(&cell_id); } - let output = self - .client - .query::(&target, None, Json(access_key_id.into())) + let lookup = async { + let observed = self + .client + .query::(&target, None, Json(access_key_id.into())) + .await; + if matches!( + &observed, + Err(InvocationError::NotStarted(Error::CellNotActive)) + ) { + // Pressure can release the owner while the peer transport paces + // a pre-dispatch refusal. Re-enter the resolver once so it can + // restore the now-idle Cell rather than forwarding to no owner. + self.client + .query::(&target, None, Json(access_key_id.into())) + .await + } else { + observed + } + }; + let output = tokio::time::timeout(QUERY_TIMEOUT, lookup) .await - .map_err(|_| internal_error())?; + .map_err(|_| { + tracing::warn!(cell = ?target.cell_id(), "credential Cell query deadline reached"); + internal_error() + })? + .map_err(|error| { + tracing::warn!(error = %error, cell = ?target.cell_id(), "credential Cell query failed"); + internal_error() + })?; output .output .0 diff --git a/src/item_wire.rs b/src/item_wire.rs new file mode 100644 index 0000000..b0f238c --- /dev/null +++ b/src/item_wire.rs @@ -0,0 +1,314 @@ +//! Compact internal read images. DynamoDB JSON and stored items are unchanged. + +use std::collections::{BTreeMap, BTreeSet}; + +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError}; +use extenddb_core::types::{AttributeValue, Item}; + +type Result = std::result::Result; +const MAX_DEPTH: usize = 32; + +fn add(left: usize, right: usize) -> Result { + left.checked_add(right).ok_or(CodecError::Limit) +} + +fn bytes_size(bytes: &[u8]) -> Result { + add(4, bytes.len()) +} + +fn collection_depth(depth: usize) -> Result { + if depth >= MAX_DEPTH { + return Err(CodecError::Invalid("item nesting exceeds 32 levels")); + } + Ok(depth + 1) +} + +fn map_size(map: &Item, depth: usize) -> Result { + map.iter().try_fold(4, |size, (name, value)| { + add( + add(size, bytes_size(name.as_bytes())?)?, + value_size(value, depth)?, + ) + }) +} + +fn value_size(value: &AttributeValue, depth: usize) -> Result { + let payload = match value { + AttributeValue::S(value) | AttributeValue::N(value) => bytes_size(value.as_bytes())?, + AttributeValue::B(value) => bytes_size(value)?, + AttributeValue::SS(values) | AttributeValue::NS(values) => values + .iter() + .try_fold(4, |size, value| add(size, bytes_size(value.as_bytes())?))?, + AttributeValue::BS(values) => values + .iter() + .try_fold(4, |size, value| add(size, bytes_size(value)?))?, + AttributeValue::Bool(_) => 1, + AttributeValue::Null => 0, + AttributeValue::L(values) => { + let next = collection_depth(depth)?; + values + .iter() + .try_fold(4, |size, value| add(size, value_size(value, next)?))? + } + AttributeValue::M(values) => map_size(values, collection_depth(depth)?)?, + }; + add(1, payload) +} + +pub(crate) fn images_size(images: &[Option]) -> Result { + if images.len() > 100 { + return Err(CodecError::Invalid("transaction read count exceeds 100")); + } + images.iter().try_fold(4, |size, item| { + add( + add(size, 1)?, + item.as_ref().map_or(Ok(0), |item| map_size(item, 0))?, + ) + }) +} + +pub(crate) fn encode_images(images: &[Option], encoder: &mut BoundedEncoder) -> Result<()> { + encoder.write_count(images.len())?; + for item in images { + encode_image(item.as_ref(), encoder)?; + } + Ok(()) +} + +pub(crate) fn encode_image(item: Option<&Item>, encoder: &mut BoundedEncoder) -> Result<()> { + encoder.write_bool(item.is_some())?; + if let Some(item) = item { + encode_map(item, encoder, 0)?; + } + Ok(()) +} + +fn encode_map(map: &Item, encoder: &mut BoundedEncoder, depth: usize) -> Result<()> { + encoder.write_count(map.len())?; + for (name, value) in map { + encoder.write_text(name)?; + encode_value(value, encoder, depth)?; + } + Ok(()) +} + +fn encode_value(value: &AttributeValue, encoder: &mut BoundedEncoder, depth: usize) -> Result<()> { + let tag = match value { + AttributeValue::S(_) => 0, + AttributeValue::N(_) => 1, + AttributeValue::B(_) => 2, + AttributeValue::SS(_) => 3, + AttributeValue::NS(_) => 4, + AttributeValue::BS(_) => 5, + AttributeValue::Bool(_) => 6, + AttributeValue::Null => 7, + AttributeValue::L(_) => 8, + AttributeValue::M(_) => 9, + }; + encoder.write_u8(tag)?; + match value { + AttributeValue::S(value) | AttributeValue::N(value) => encoder.write_text(value), + AttributeValue::B(value) => encoder.write_bytes(value), + AttributeValue::SS(values) | AttributeValue::NS(values) => { + encoder.write_count(values.len())?; + for value in values { + encoder.write_text(value)?; + } + Ok(()) + } + AttributeValue::BS(values) => { + encoder.write_count(values.len())?; + for value in values { + encoder.write_bytes(value)?; + } + Ok(()) + } + AttributeValue::Bool(value) => encoder.write_bool(*value), + AttributeValue::Null => Ok(()), + AttributeValue::L(values) => { + let next = collection_depth(depth)?; + encoder.write_count(values.len())?; + for value in values { + encode_value(value, encoder, next)?; + } + Ok(()) + } + AttributeValue::M(values) => encode_map(values, encoder, collection_depth(depth)?), + } +} + +pub(crate) fn decode_images(decoder: &mut BoundedDecoder<'_>) -> Result>> { + let count = decoder.read_count()?; + if count > 100 { + return Err(CodecError::Invalid("transaction read count exceeds 100")); + } + let mut images = Vec::with_capacity(count); + for _ in 0..count { + images.push(if decoder.read_bool()? { + Some(decode_map(decoder, 0)?) + } else { + None + }); + } + Ok(images) +} + +fn decode_map(decoder: &mut BoundedDecoder<'_>, depth: usize) -> Result { + let count = decoder.read_count()?; + let mut map: Item = BTreeMap::new(); + for _ in 0..count { + let name = decoder.read_text()?; + if map + .last_key_value() + .is_some_and(|(previous, _)| previous.as_str() >= name) + { + return Err(CodecError::Invalid("item keys are not strictly ordered")); + } + let value = decode_value(decoder, depth)?; + map.insert(name.to_owned(), value); + } + Ok(map) +} + +fn decode_set( + decoder: &mut BoundedDecoder<'_>, + read: impl Fn(&mut BoundedDecoder<'_>) -> Result, +) -> Result> { + let count = decoder.read_count()?; + if count == 0 { + return Err(CodecError::Invalid("empty attribute set")); + } + let mut set = BTreeSet::new(); + for _ in 0..count { + let value = read(decoder)?; + if set.last().is_some_and(|previous| previous >= &value) { + return Err(CodecError::Invalid("attribute set is not strictly ordered")); + } + set.insert(value); + } + Ok(set) +} + +fn decode_value(decoder: &mut BoundedDecoder<'_>, depth: usize) -> Result { + Ok(match decoder.read_u8()? { + 0 => AttributeValue::S(decoder.read_text()?.to_owned()), + 1 => AttributeValue::N(decoder.read_text()?.to_owned()), + 2 => AttributeValue::B(decoder.read_bytes()?.to_vec()), + 3 => AttributeValue::SS(decode_set(decoder, |decoder| { + Ok(decoder.read_text()?.to_owned()) + })?), + 4 => AttributeValue::NS(decode_set(decoder, |decoder| { + Ok(decoder.read_text()?.to_owned()) + })?), + 5 => AttributeValue::BS(decode_set(decoder, |decoder| { + Ok(decoder.read_bytes()?.to_vec()) + })?), + 6 => AttributeValue::Bool(decoder.read_bool()?), + 7 => AttributeValue::Null, + 8 => { + let next = collection_depth(depth)?; + let count = decoder.read_count()?; + // Do not preallocate from a peer-supplied count. Truncation is + // checked as each element is consumed from the bounded decoder. + let mut values = Vec::new(); + for _ in 0..count { + values.push(decode_value(decoder, next)?); + } + AttributeValue::L(values) + } + 9 => AttributeValue::M(decode_map(decoder, collection_depth(depth)?)?), + _ => return Err(CodecError::Invalid("unknown attribute tag")), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn decode(bytes: &[u8]) -> Result>> { + let mut decoder = BoundedDecoder::new(bytes, crate::OPERATION_BYTES)?; + let value = decode_images(&mut decoder)?; + decoder.finish()?; + Ok(value) + } + + #[test] + fn malformed_images_are_rejected_without_allocating_peer_counts() { + for bytes in [ + vec![0xff; 4], + vec![0, 0, 0, 1, 2], + vec![0, 0, 0, 1, 1, 0xff, 0xff, 0xff, 0xff], + ] { + assert!(decode(&bytes).is_err()); + } + let mut encoded = BoundedEncoder::new(1024).unwrap(); + encode_images( + &[Some(Item::from([("name".into(), AttributeValue::Null)]))], + &mut encoded, + ) + .unwrap(); + let bytes = encoded.finish(); + for length in 0..bytes.len() { + assert!(decode(&bytes[..length]).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(decode(&trailing).is_err()); + let mut unknown = bytes; + *unknown.last_mut().unwrap() = 255; + assert!(decode(&unknown).is_err()); + } + + #[test] + fn compact_maps_and_sets_require_canonical_order() { + for is_set in [false, true] { + let mut encoder = BoundedEncoder::new(1024).unwrap(); + encoder.write_count(1).unwrap(); + encoder.write_bool(true).unwrap(); + encoder.write_count(if is_set { 1 } else { 2 }).unwrap(); + encoder.write_text("b").unwrap(); + if is_set { + encoder.write_u8(3).unwrap(); + encoder.write_count(2).unwrap(); + encoder.write_text("b").unwrap(); + encoder.write_text("a").unwrap(); + } else { + encoder.write_u8(7).unwrap(); + encoder.write_text("a").unwrap(); + encoder.write_u8(7).unwrap(); + } + assert!(decode(&encoder.finish()).is_err()); + } + } + + #[test] + fn compact_values_enforce_the_dynamodb_nesting_limit() { + let mut value = AttributeValue::Null; + for _ in 0..32 { + value = AttributeValue::L(vec![value]); + } + let images = [Some(Item::from([("nested".into(), value.clone())]))]; + let size = images_size(&images).unwrap(); + let mut encoder = BoundedEncoder::new(1024).unwrap(); + encode_images(&images, &mut encoder).unwrap(); + let bytes = encoder.finish(); + assert_eq!(size, bytes.len()); + assert!(decode(&bytes).unwrap() == images); + let invalid = [Some(Item::from([( + "nested".into(), + AttributeValue::L(vec![value]), + )]))]; + assert!(images_size(&invalid).is_err()); + let mut malicious = BoundedEncoder::new(1024).unwrap(); + malicious.write_count(1).unwrap(); + malicious.write_bool(true).unwrap(); + malicious.write_count(1).unwrap(); + malicious.write_text("nested").unwrap(); + for _ in 0..33 { + malicious.write_u8(8).unwrap(); + malicious.write_count(1).unwrap(); + } + malicious.write_u8(7).unwrap(); + assert!(decode(&malicious.finish()).is_err()); + } +} diff --git a/src/items.rs b/src/items.rs index ff8a07e..34f1521 100644 --- a/src/items.rs +++ b/src/items.rs @@ -576,9 +576,10 @@ impl Command for TransactWriteNoReturn { pub(crate) mod transaction; pub use transaction::{ - PrepareAccountTransaction, PrepareAccountTransactionInput, ReadAccountTransaction, - ReadAccountTransactionResult, ReleaseAccountTransactionReads, ResolveAccountTransaction, - TransactRead, TransactReadQuery, TransactionReadOutcome, + PrepareAccountTransaction, PrepareAccountTransactionBounded, PrepareAccountTransactionInput, + ReadAccountTransaction, ReadAccountTransactionResult, ReadAccountTransactionResultBounded, + ReleaseAccountTransactionReads, ResolveAccountTransaction, TransactRead, TransactReadQuery, + TransactionReadOutcome, TransactionReadQueryOutput, }; mod scan; diff --git a/src/items/transaction.rs b/src/items/transaction.rs index aad60de..c6be5b6 100644 --- a/src/items/transaction.rs +++ b/src/items/transaction.rs @@ -210,6 +210,8 @@ pub(super) fn without_old_image(reason: TransactionFailure) -> TransactionFailur pub enum TransactionReadOutcome { /// Every requested image was read from one Cell snapshot. Applied(Vec>), + /// The encoded aggregate requires reading durable participant images individually. + SavedImagesRequired, /// No image was returned because one operation failed validation or locking. Rejected { /// Position of the failing read. @@ -254,21 +256,57 @@ impl Command for TransactRead { /// serialized with commands while avoiding a durable mutation publication. pub struct TransactReadQuery; +/// Compact read images with an explicit fallback for oversized aggregates. +#[derive(Clone, Debug, PartialEq)] +pub struct TransactionReadQueryOutput(pub TransactionReadOutcome); + +impl WireValue for TransactionReadQueryOutput { + fn encode(&self, encoder: &mut BoundedEncoder) -> std::result::Result<(), CodecError> { + encode_read_query( + &self.0, + match &self.0 { + TransactionReadOutcome::Applied(images) => Some(images), + _ => None, + }, + &TransactionReadOutcome::SavedImagesRequired, + encoder, + ) + } + + fn decode(decoder: &mut BoundedDecoder<'_>) -> std::result::Result { + Ok(Self( + if let Some(images) = crate::decode_read_query_images(decoder)? { + TransactionReadOutcome::Applied(images) + } else { + let outcome = Json::::decode(decoder)?.0; + if matches!(outcome, TransactionReadOutcome::Applied(_)) { + return Err(CodecError::Invalid("read images require compact envelope")); + } + outcome + }, + )) + } +} + impl Query for TransactReadQuery { const MODULE: &'static str = MODULE; const ID: u32 = 54; - const CODEC_VERSION: u32 = 1; + const CODEC_VERSION: u32 = 3; type Input = Json; - type Output = Json; + type Output = TransactionReadQueryOutput; fn execute(context: &mut QueryContext<'_>, Json(input): Self::Input) -> Result { let images = match query_stage(context, input.operations)? { Ok(images) => images, Err((index, reason)) => { - return Ok(Json(TransactionReadOutcome::Rejected { index, reason })); + return Ok(TransactionReadQueryOutput( + TransactionReadOutcome::Rejected { index, reason }, + )); } }; - Ok(Json(TransactionReadOutcome::Applied(images))) + Ok(TransactionReadQueryOutput(TransactionReadOutcome::Applied( + images, + ))) } } @@ -354,51 +392,78 @@ impl Command for PrepareAccountTransaction { Json(input): Self::Input, ) -> Result> { let input = crate::transaction_transport::consume::(context, input)?; - let digest = blake3::hash(&serde_json::to_vec(&input)?); - if let Some(outcome) = participant::prepared( - context, - input.transaction_id, - input.coordinator_cell, - digest, - )? { - return Ok(CommandResult::Rejected(Json(outcome))); + prepare_account(context, input) + } +} + +/// Prepare an inline participant with a bounded reply reservation. +/// A durable WideRequired rejection permits retry through the wide command. +pub struct PrepareAccountTransactionBounded; + +impl Command for PrepareAccountTransactionBounded { + const MODULE: &'static str = MODULE; + const ID: u32 = 57; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + let result = prepare_account(context, input)?; + crate::participant::bound_prepare_result(result) + } +} + +fn prepare_account( + context: &mut CommandContext<'_, '_>, + input: PrepareAccountTransactionInput, +) -> Result>> { + let digest = blake3::hash(&serde_json::to_vec(&input)?); + if let Some(outcome) = participant::prepared( + context, + input.transaction_id, + input.coordinator_cell, + digest, + )? { + return Ok(CommandResult::Rejected(Json(outcome))); + } + let staged = match stage(context, input.operations)? { + Ok(staged) => staged, + Err((index, reason)) => { + return Ok(CommandResult::Rejected(Json( + PrepareTransactionOutcome::Rejected { index, reason }, + ))); } - let staged = match stage(context, input.operations)? { - Ok(staged) => staged, - Err((index, reason)) => { - return Ok(CommandResult::Rejected(Json( - PrepareTransactionOutcome::Rejected { index, reason }, - ))); - } - }; - participant::record_prepare( - context, - input.transaction_id, - input.coordinator_cell, - digest, - crate::participant::PreparedPayload { - bytes: serde_json::to_vec(&staged)?, - operations: staged.len(), - index_edits: staged.iter().map(|image| image.index_capacity.edits).sum(), - index_overflow_bytes: staged - .iter() - .map(|image| image.index_capacity.overflow_bytes) - .sum(), - }, - &input.coordinator_key, - staged + }; + participant::record_prepare( + context, + input.transaction_id, + input.coordinator_cell, + digest, + crate::participant::PreparedPayload { + bytes: serde_json::to_vec(&staged)?, + operations: staged.len(), + index_edits: staged.iter().map(|image| image.index_capacity.edits).sum(), + index_overflow_bytes: staged .iter() - .filter(|image| image.effect == StagedEffect::Read) - .map(|image| image.image.as_ref()), - )?; - for image in staged { - context.sql(&statement("INSERT OR IGNORE INTO ddb_account_transaction_locks (table_id, item_key, transaction_id, write_lock) VALUES (?1, ?2, ?3, ?4)", - vec![SqlValue::Text(image.table_id), SqlValue::Blob(image.key), SqlValue::Blob(input.transaction_id.to_vec()), SqlValue::Integer(i64::from(image.effect != StagedEffect::Read))]))?; - } - Ok(CommandResult::Success(Json( - PrepareTransactionOutcome::Prepared, - ))) + .map(|image| image.index_capacity.overflow_bytes) + .sum(), + }, + &input.coordinator_key, + staged + .iter() + .filter(|image| image.effect == StagedEffect::Read) + .map(|image| image.image.as_ref()), + )?; + for image in staged { + context.sql(&statement("INSERT OR IGNORE INTO ddb_account_transaction_locks (table_id, item_key, transaction_id, write_lock) VALUES (?1, ?2, ?3, ?4)", + vec![SqlValue::Text(image.table_id), SqlValue::Blob(image.key), SqlValue::Blob(input.transaction_id.to_vec()), SqlValue::Integer(i64::from(image.effect != StagedEffect::Read))]))?; } + Ok(CommandResult::Success(Json( + PrepareTransactionOutcome::Prepared, + ))) } /// Apply a coordinator decision and release the account participant's locks atomically. @@ -497,6 +562,19 @@ impl Query for ReadAccountTransactionResult { } } +/// Read a saved account image without reserving the wide item reply budget. +pub struct ReadAccountTransactionResultBounded; +impl Query for ReadAccountTransactionResultBounded { + const MODULE: &'static str = MODULE; + const ID: u32 = 56; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = crate::BoundedTransactionReadResult; + fn execute(context: &mut QueryContext<'_>, Json(input): Self::Input) -> Result { + crate::participant::bounded_read_result(context, input) + } +} + /// Release assembled read images while retaining the account's terminal decision. pub struct ReleaseAccountTransactionReads; impl Command for ReleaseAccountTransactionReads { diff --git a/src/lib.rs b/src/lib.rs index aea7921..86f1632 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -8,6 +8,7 @@ mod directory; mod expression_wire; mod global_index; mod item_storage; +mod item_wire; mod items; mod participant; mod partition; @@ -32,17 +33,17 @@ pub use expression_wire::WireCondition; pub use global_index::*; pub use items::*; pub use participant::{ - ParticipantTransactionState, PrepareTransactionOutcome, ReadTransactionInput, - ReadTransactionResultInput, ResolveTransactionInput, ResolveTransactionOutcome, - TransactionReadConflict, TransactionReadResult, + BoundedTransactionReadResult, ParticipantTransactionState, PrepareTransactionOutcome, + ReadTransactionInput, ReadTransactionResultInput, ResolveTransactionInput, + ResolveTransactionOutcome, TransactionReadConflict, TransactionReadResult, }; pub use partition::*; pub use provision::*; pub use routing::*; pub use server::{ BeyonddbPeerScope, BeyonddbPeers, NodeLeasePublisher, PeerNodeDurabilityProvider, - PeerNodeLogTransport, PublishedNodeLease, PublishedNodeLogAuthority, build_http_state, - build_http_state_with_cache, measured_node_capacity, recover_fenced_node_log, + PeerNodeLogTransport, PublishedNodeLease, PublishedNodeLogAuthority, RuntimeMetrics, + build_http_state, build_http_state_with_cache, measured_node_capacity, recover_fenced_node_log, shutdown_serving_node, }; pub use split::*; @@ -139,6 +140,15 @@ const fn operation(id: u32) -> OperationDescriptor { } } +const fn coordinator_registration_operation(id: u32) -> OperationDescriptor { + OperationDescriptor { + // Only an account ID (at most 128 bytes) and shard numbers; no images. + input_limit: 4096, + output_limit: 4096, + ..operation(id) + } +} + const fn no_return_operation(id: u32) -> OperationDescriptor { OperationDescriptor { id, @@ -172,7 +182,7 @@ const fn no_return_transaction_operation(id: u32) -> OperationDescriptor { } } -static COMMANDS: [OperationDescriptor; 33] = [ +static COMMANDS: [OperationDescriptor; 35] = [ operation(1), operation(2), operation(3), @@ -190,7 +200,7 @@ static COMMANDS: [OperationDescriptor; 33] = [ operation(17), operation(18), operation(19), - operation(20), + coordinator_registration_operation(20), operation(21), participant::phase_operation(22), crate::transaction_transport::upload_operation(23), @@ -225,8 +235,10 @@ static COMMANDS: [OperationDescriptor; 33] = [ operation(52), no_return_transaction_operation(53), no_return_operation(55), + coordinator_registration_operation(56), + participant::bounded_prepare_operation(57), ]; -static QUERIES: [OperationDescriptor; 32] = [ +static QUERIES: [OperationDescriptor; 33] = [ operation(4), operation(7), OperationDescriptor { @@ -251,7 +263,7 @@ static QUERIES: [OperationDescriptor; 32] = [ ..operation(24) }, participant::phase_operation(25), - operation(26), + coordinator_registration_operation(26), operation(27), operation(28), OperationDescriptor { @@ -276,7 +288,11 @@ static QUERIES: [OperationDescriptor; 32] = [ operation(47), operation(48), operation(49), - operation(54), + OperationDescriptor { + codec_version: 3, + ..operation(54) + }, + participant::bounded_read_result_operation(56), ]; /// Statically linked account application. @@ -458,6 +474,7 @@ impl cellule_runtime::registry::CellModule for AccountModule { registry.bind_command::()?; registry.bind_command::>()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; @@ -479,6 +496,7 @@ impl cellule_runtime::registry::CellModule for AccountModule { registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; @@ -486,6 +504,7 @@ impl cellule_runtime::registry::CellModule for AccountModule { registry.bind_query::()?; registry.bind_query::()?; registry.bind_query::()?; + registry.bind_query::()?; registry.bind_query::()?; registry.bind_query::()?; registry.bind_query::()?; @@ -510,6 +529,48 @@ impl cellule_runtime::registry::CellModule for AccountModule { #[derive(Clone, Debug, PartialEq)] pub struct Json(pub T); +// Compact images avoid JSON binary/escape expansion across peer hops. Keep an +// explicit saved-image fallback when even the compact aggregate exceeds the +// operation envelope; non-item outcomes retain canonical JSON. +fn encode_read_query( + value: &T, + images: Option<&[Option]>, + fallback: &T, + encoder: &mut BoundedEncoder, +) -> std::result::Result<(), CodecError> { + let json = if let Some(images) = images { + match item_wire::images_size(images) { + Ok(size) if size < OPERATION_BYTES as usize => { + encoder.write_u8(1)?; + return item_wire::encode_images(images, encoder); + } + Ok(_) | Err(CodecError::Limit) => fallback, + Err(error) => return Err(error), + } + } else { + value + }; + let mut bytes = serde_json::to_vec(json) + .map_err(|_| CodecError::Invalid("DynamoDB value failed to encode"))?; + // The JSON envelope has a tag and a four-byte length prefix. + if bytes.len() > OPERATION_BYTES as usize - 5 { + bytes = serde_json::to_vec(fallback) + .map_err(|_| CodecError::Invalid("DynamoDB value failed to encode"))?; + } + encoder.write_u8(0)?; + encoder.write_bytes(&bytes) +} + +fn decode_read_query_images( + decoder: &mut BoundedDecoder<'_>, +) -> std::result::Result>>, CodecError> { + match decoder.read_u8()? { + 0 => Ok(None), + 1 => item_wire::decode_images(decoder).map(Some), + _ => Err(CodecError::Invalid("unknown transaction read envelope")), + } +} + impl WireValue for Json where T: Serialize + DeserializeOwned + Send + 'static, @@ -532,3 +593,160 @@ where Ok(Self(value)) } } + +#[cfg(test)] +mod transaction_read_codec_tests { + use super::*; + use extenddb_core::types::AttributeValue; + + fn roundtrip(value: T) -> T { + let mut encoder = BoundedEncoder::new(OPERATION_BYTES).unwrap(); + value.encode(&mut encoder).unwrap(); + let bytes = encoder.finish(); + let mut decoder = BoundedDecoder::new(&bytes, OPERATION_BYTES).unwrap(); + let result = T::decode(&mut decoder).unwrap(); + decoder.finish().unwrap(); + result + } + + #[test] + fn compact_read_preserves_nested_attributes_and_all_outcomes() { + use std::collections::BTreeSet; + let attributes = Item::from([ + ( + "string".into(), + AttributeValue::S(['\0', '\n', '\\', '"', '界'].into_iter().collect()), + ), + ("number".into(), AttributeValue::N("-123.456".into())), + ("binary".into(), AttributeValue::B(vec![0, 128, 255])), + ( + "strings".into(), + AttributeValue::SS(BTreeSet::from(["a".into(), "界".into()])), + ), + ( + "numbers".into(), + AttributeValue::NS(BTreeSet::from([ + "-2".into(), + "10000000000000000000000000000000000000".into(), + ])), + ), + ( + "binaries".into(), + AttributeValue::BS(BTreeSet::from([vec![0], vec![255]])), + ), + ("boolean".into(), AttributeValue::Bool(true)), + ("null".into(), AttributeValue::Null), + ( + "list".into(), + AttributeValue::L(vec![ + AttributeValue::Bool(false), + AttributeValue::M(Item::from([("nested".into(), AttributeValue::B(vec![]))])), + ]), + ), + ( + "map".into(), + AttributeValue::M(Item::from([("value".into(), AttributeValue::S("".into()))])), + ), + ]); + let images = vec![Some(attributes), None, Some(Item::new())]; + let account = TransactionReadOutcome::Applied(images.clone()); + let partition = PartitionTransactReadOutcome::Applied(images); + assert_eq!( + roundtrip(TransactionReadQueryOutput(account.clone())).0, + account + ); + assert_eq!( + roundtrip(PartitionTransactReadQueryOutput(partition.clone())).0, + partition + ); + for outcome in [ + PartitionTransactReadOutcome::NotInstalled, + PartitionTransactReadOutcome::StaleRoute, + PartitionTransactReadOutcome::Sealed, + PartitionTransactReadOutcome::NotReady, + PartitionTransactReadOutcome::WrongPartition, + PartitionTransactReadOutcome::SavedImagesRequired, + PartitionTransactReadOutcome::Rejected { + index: 3, + reason: TransactionFailure::Conflict, + }, + ] { + assert_eq!( + roundtrip(PartitionTransactReadQueryOutput(outcome.clone())).0, + outcome + ); + } + let rejected = TransactionReadOutcome::Rejected { + index: 2, + reason: TransactionFailure::Validation("invalid read".into()), + }; + assert_eq!( + roundtrip(TransactionReadQueryOutput(rejected.clone())).0, + rejected + ); + } + + #[test] + fn read_images_reject_noncanonical_json_envelopes() { + let mut encoder = BoundedEncoder::new(OPERATION_BYTES).unwrap(); + encoder.write_u8(0).unwrap(); + Json(TransactionReadOutcome::Applied(vec![])) + .encode(&mut encoder) + .unwrap(); + let bytes = encoder.finish(); + let mut decoder = BoundedDecoder::new(&bytes, OPERATION_BYTES).unwrap(); + assert!(TransactionReadQueryOutput::decode(&mut decoder).is_err()); + let mut decoder = BoundedDecoder::new(&bytes, OPERATION_BYTES).unwrap(); + assert!(PartitionTransactReadQueryOutput::decode(&mut decoder).is_err()); + } + + #[test] + fn compact_read_keeps_large_escaped_strings_inside_the_query_envelope() { + let images = vec![ + Some(Item::from([( + "payload".into(), + AttributeValue::S("\0".repeat(380 * 1024)) + )])); + 10 + ]; + let value = TransactionReadOutcome::Applied(images); + assert!(roundtrip(TransactionReadQueryOutput(value.clone())).0 == value); + } + + #[test] + fn transaction_read_codecs_preserve_legal_aggregates_and_bound_output() { + for (count, size, oversized) in [ + (1, 100, false), + (10, 380 * 1024, false), + (12, 380 * 1024, true), + ] { + let images = vec![ + Some(Item::from([( + "payload".into(), + AttributeValue::B(vec![0xa5; size]), + )])); + count + ]; + let account = TransactionReadOutcome::Applied(images.clone()); + let partition = PartitionTransactReadOutcome::Applied(images); + let account_result = roundtrip(TransactionReadQueryOutput(account.clone())).0; + let partition_result = roundtrip(PartitionTransactReadQueryOutput(partition.clone())).0; + if oversized { + assert_eq!(account_result, TransactionReadOutcome::SavedImagesRequired); + assert_eq!( + partition_result, + PartitionTransactReadOutcome::SavedImagesRequired + ); + } else { + assert!( + account_result == account, + "legal account aggregate must retain all items" + ); + assert!( + partition_result == partition, + "legal partition aggregate must retain all items" + ); + } + } + } +} diff --git a/src/participant.rs b/src/participant.rs index 2e9415f..7ae2fca 100644 --- a/src/participant.rs +++ b/src/participant.rs @@ -2,6 +2,7 @@ use crate::table::statement; use crate::{Error, Json, Result, SqlValue, TransactionFailure}; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; use cellule_runtime::registry::{CommandContext, CommandResult, QueryContext}; use extenddb_core::types::Item; use serde::{Deserialize, Serialize}; @@ -22,6 +23,40 @@ pub(crate) const fn phase_operation(id: u32) -> cellule_runtime::registry::Opera } } +// The adapter selects this path for payloads at most 32 KiB. The larger input +// envelope leaves codec headroom and the result budget admits concurrent small +// prepares without reserving 4 MiB per request in the 16 MiB Cell mailbox. +pub(crate) const SMALL_PREPARE_BYTES: usize = 32 * 1024; +const PREPARE_BYTES: u32 = 64 * 1024; + +pub(crate) const fn bounded_prepare_operation( + id: u32, +) -> cellule_runtime::registry::OperationDescriptor { + cellule_runtime::registry::OperationDescriptor { + input_limit: PREPARE_BYTES, + output_limit: PREPARE_BYTES, + ..crate::operation(id) + } +} + +pub(crate) fn bound_prepare_result( + result: CommandResult>, +) -> Result>> { + let output = match &result { + CommandResult::Success(output) | CommandResult::Rejected(output) => output, + }; + let mut encoder = BoundedEncoder::new(PREPARE_BYTES)?; + match output.encode(&mut encoder) { + Ok(()) => Ok(result), + // Cellule rolls back application changes before durably recording this + // rejected receipt. Do not truncate a condition failure's old image. + Err(CodecError::Limit) => Ok(CommandResult::Rejected(Json( + PrepareTransactionOutcome::WideRequired, + ))), + Err(error) => Err(error.into()), + } +} + #[derive(Clone, Copy, Debug, PartialEq, Serialize, Deserialize)] pub(crate) enum StagedEffect { Write, @@ -46,6 +81,9 @@ pub enum PrepareTransactionOutcome { Sealed, NotReady, WrongPartition, + /// Bounded prepare rolled back; retry with the wide reply envelope. + /// Emitted only by the bounded prepare commands. + WideRequired, } /// A terminal coordinator decision for one participant. @@ -370,6 +408,77 @@ pub enum TransactionReadResult { Item(Option), } +const READ_RESULT_BYTES: u32 = 64 * 1024; + +pub(crate) const fn bounded_read_result_operation( + id: u32, +) -> cellule_runtime::registry::OperationDescriptor { + cellule_runtime::registry::OperationDescriptor { + input_limit: 4096, + output_limit: READ_RESULT_BYTES, + ..crate::operation(id) + } +} + +/// Compact immutable participant image with an explicit wide-query fallback. +#[derive(Clone, Debug, PartialEq)] +pub enum BoundedTransactionReadResult { + /// No committed saved image exists for this identity and position. + Unavailable, + /// The complete saved image, including an absent item. + Item(Option), + /// Fetch the same saved image through the existing wide query. + WideRequired, +} + +impl WireValue for BoundedTransactionReadResult { + fn encode(&self, encoder: &mut BoundedEncoder) -> std::result::Result<(), CodecError> { + match self { + Self::Unavailable => encoder.write_u8(0), + Self::WideRequired => encoder.write_u8(2), + Self::Item(item) => { + let images = std::slice::from_ref(item); + match crate::item_wire::images_size(images) { + Ok(size) if size < READ_RESULT_BYTES as usize => { + encoder.write_u8(1)?; + crate::item_wire::encode_images(images, encoder) + } + Ok(_) | Err(CodecError::Limit) => encoder.write_u8(2), + Err(error) => Err(error), + } + } + } + } + + fn decode(decoder: &mut BoundedDecoder<'_>) -> std::result::Result { + match decoder.read_u8()? { + 0 => Ok(Self::Unavailable), + 1 => { + let mut images = crate::item_wire::decode_images(decoder)?; + if images.len() != 1 { + return Err(CodecError::Invalid("saved read requires one image")); + } + images + .pop() + .map(Self::Item) + .ok_or(CodecError::Invalid("missing saved image")) + } + 2 => Ok(Self::WideRequired), + _ => Err(CodecError::Invalid("unknown saved read envelope")), + } + } +} + +pub(crate) fn bounded_read_result( + context: &QueryContext<'_>, + input: ReadTransactionResultInput, +) -> Result { + Ok(match read_result(context, input)?.0 { + TransactionReadResult::Unavailable => BoundedTransactionReadResult::Unavailable, + TransactionReadResult::Item(item) => BoundedTransactionReadResult::Item(item), + }) +} + pub(crate) fn read_result( context: &QueryContext<'_>, input: ReadTransactionResultInput, @@ -442,3 +551,63 @@ pub(crate) fn read( }; Ok(Json(outcome)) } + +#[cfg(test)] +mod saved_read_tests { + use super::*; + use extenddb_core::types::AttributeValue; + + fn encode(value: &BoundedTransactionReadResult) -> Vec { + let mut encoder = BoundedEncoder::new(READ_RESULT_BYTES).unwrap(); + value.encode(&mut encoder).unwrap(); + encoder.finish() + } + + fn decode(bytes: &[u8]) -> std::result::Result { + let mut decoder = BoundedDecoder::new(bytes, READ_RESULT_BYTES)?; + let value = BoundedTransactionReadResult::decode(&mut decoder)?; + decoder.finish()?; + Ok(value) + } + + #[test] + fn saved_read_exact_limit_preserves_image_and_larger_reply_requests_fallback() { + let image = |size| { + BoundedTransactionReadResult::Item(Some(Item::from([( + "value".into(), + AttributeValue::B(vec![0xa5; size]), + )]))) + }; + let largest = image(65_512); + let bytes = encode(&largest); + assert_eq!(bytes.len(), READ_RESULT_BYTES as usize); + assert_eq!(decode(&bytes).unwrap(), largest); + assert_eq!(encode(&image(65_513)), vec![2]); + assert_eq!( + decode(&[2]).unwrap(), + BoundedTransactionReadResult::WideRequired + ); + } + + #[test] + fn saved_read_rejects_malformed_envelopes_and_preserves_absent_item() { + for count in [0, 2] { + let mut encoder = BoundedEncoder::new(64).unwrap(); + encoder.write_u8(1).unwrap(); + encoder.write_count(count).unwrap(); + for _ in 0..count { + encoder.write_bool(false).unwrap(); + } + assert!(decode(&encoder.finish()).is_err()); + } + for bytes in [&[255][..], &[2, 0][..], &[1][..]] { + assert!(decode(bytes).is_err()); + } + let absent = BoundedTransactionReadResult::Item(None); + assert_eq!(decode(&encode(&absent)).unwrap(), absent); + assert_ne!( + encode(&absent), + encode(&BoundedTransactionReadResult::Unavailable) + ); + } +} diff --git a/src/partition.rs b/src/partition.rs index 5770ca7..ce567aa 100644 --- a/src/partition.rs +++ b/src/partition.rs @@ -6,6 +6,7 @@ pub(crate) mod query; mod scan; mod transaction; mod ttl; +mod update_batch; pub use indexes::*; pub use key::data_key_hash; @@ -13,6 +14,7 @@ pub use query::*; pub use scan::*; pub use transaction::*; pub use ttl::*; +pub use update_batch::*; use std::sync::OnceLock; @@ -53,7 +55,7 @@ static NAMESPACES: [NamespaceDescriptor; 1] = [NamespaceDescriptor { effect_targets: &[], dead_letter: None, }]; -static COMMANDS: [OperationDescriptor; 25] = [ +static COMMANDS: [OperationDescriptor; 29] = [ operation(1), operation(2), OperationDescriptor { @@ -84,12 +86,16 @@ static COMMANDS: [OperationDescriptor; 25] = [ no_return_operation(22), operation(23), crate::no_return_transaction_operation(24), + update_batch::batch_operation(25), + operation(26), + crate::participant::bounded_prepare_operation(28), + crate::participant::bounded_prepare_operation(29), OperationDescriptor { codec_version: 2, ..no_return_operation(52) }, ]; -static QUERIES: [OperationDescriptor; 18] = [ +static QUERIES: [OperationDescriptor; 19] = [ operation(1), operation(2), operation(3), @@ -107,7 +113,11 @@ static QUERIES: [OperationDescriptor; 18] = [ operation(16), operation(17), operation(18), - operation(19), + OperationDescriptor { + codec_version: 3, + ..operation(19) + }, + crate::participant::bounded_read_result_operation(20), ]; const fn operation(id: u32) -> OperationDescriptor { @@ -155,7 +165,10 @@ impl cellule_runtime::registry::CellModule for DataModule { source.update(include_bytes!("partition/scan.rs")); source.update(include_bytes!("partition/transaction.rs")); source.update(include_bytes!("partition/transaction/participant.rs")); + source.update(include_bytes!("partition/transaction/prepare_batch.rs")); source.update(include_bytes!("partition/ttl.rs")); + source.update(include_bytes!("partition/update_batch.rs")); + source.update(include_bytes!("item_wire.rs")); source.update(include_bytes!("partition/indexes.rs")); source.update(include_bytes!("items.rs")); source.update(include_bytes!("item_storage.rs")); @@ -198,6 +211,8 @@ impl cellule_runtime::registry::CellModule for DataModule { registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; @@ -210,6 +225,8 @@ impl cellule_runtime::registry::CellModule for DataModule { registry.bind_command::()?; registry.bind_command::>()?; registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; @@ -227,7 +244,8 @@ impl cellule_runtime::registry::CellModule for DataModule { registry.bind_query::()?; registry.bind_query::()?; registry.bind_query::()?; - registry.bind_query::() + registry.bind_query::()?; + registry.bind_query::() } } @@ -627,7 +645,7 @@ impl Command for SealPartition { SealPartitionOutcome::StaleRoute, ))); } - if !seal.valid_for(&source) { + if source.table.placement == crate::TablePlacement::Single || !seal.valid_for(&source) { return Ok(CommandResult::Rejected(Json( SealPartitionOutcome::InvalidSeal, ))); @@ -1324,7 +1342,7 @@ impl Command for PartitionUpdate { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - execute_partition_update(context, input, true) + execute_partition_update(context, input, true, 0) } } @@ -1342,7 +1360,7 @@ impl Command for PartitionUpdateNoReturn { context: &mut CommandContext<'_, '_>, Json(input): Self::Input, ) -> Result> { - execute_partition_update(context, input, false) + execute_partition_update(context, input, false, 0) } } @@ -1350,6 +1368,7 @@ fn execute_partition_update( context: &mut CommandContext<'_, '_>, input: PartitionUpdateInput, return_images: bool, + ordinal: usize, ) -> Result>> { let Some(spec) = indexes::command_spec(context)? else { return Ok(CommandResult::Rejected(Json( @@ -1437,7 +1456,7 @@ fn execute_partition_update( spec.table.stream.as_ref(), old.as_ref(), Some(&new), - 0, + ordinal, )?; Ok(CommandResult::Success(Json(if return_images { PartitionUpdateOutcome::Applied { old, new } diff --git a/src/partition/transaction.rs b/src/partition/transaction.rs index 512dade..b7f17a1 100644 --- a/src/partition/transaction.rs +++ b/src/partition/transaction.rs @@ -3,6 +3,7 @@ use crate::participant::StagedEffect; use std::collections::HashSet; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; use cellule_runtime::registry::{Command, CommandContext, CommandResult, Query, QueryContext}; use extenddb_core::types::Item; use serde::{Deserialize, Serialize}; @@ -123,6 +124,8 @@ fn execute_write( pub enum PartitionTransactReadOutcome { /// Every requested image was read from one Cell snapshot. Applied(Vec>), + /// The encoded aggregate requires reading durable participant images individually. + SavedImagesRequired, /// No image was returned because one operation failed validation or locking. Rejected { /// Position of the failing read. @@ -201,12 +204,44 @@ impl Command for PartitionTransactRead { /// Read a transaction batch through Cellule's read-only query path. pub struct PartitionTransactReadQuery; +/// Compact read images with an explicit fallback for oversized aggregates. +#[derive(Clone, Debug, PartialEq)] +pub struct PartitionTransactReadQueryOutput(pub PartitionTransactReadOutcome); + +impl WireValue for PartitionTransactReadQueryOutput { + fn encode(&self, encoder: &mut BoundedEncoder) -> std::result::Result<(), CodecError> { + crate::encode_read_query( + &self.0, + match &self.0 { + PartitionTransactReadOutcome::Applied(images) => Some(images), + _ => None, + }, + &PartitionTransactReadOutcome::SavedImagesRequired, + encoder, + ) + } + + fn decode(decoder: &mut BoundedDecoder<'_>) -> std::result::Result { + Ok(Self( + if let Some(images) = crate::decode_read_query_images(decoder)? { + PartitionTransactReadOutcome::Applied(images) + } else { + let outcome = Json::::decode(decoder)?.0; + if matches!(outcome, PartitionTransactReadOutcome::Applied(_)) { + return Err(CodecError::Invalid("read images require compact envelope")); + } + outcome + }, + )) + } +} + impl Query for PartitionTransactReadQuery { const MODULE: &'static str = DATA_MODULE; const ID: u32 = 19; - const CODEC_VERSION: u32 = 1; + const CODEC_VERSION: u32 = 3; type Input = Json; - type Output = Json; + type Output = PartitionTransactReadQueryOutput; fn execute(context: &mut QueryContext<'_>, Json(input): Self::Input) -> Result { let rows = context.sql(&statement( @@ -214,28 +249,42 @@ impl Query for PartitionTransactReadQuery { vec![], ))?; let Some(spec) = super::decode_spec(&rows[0])? else { - return Ok(Json(PartitionTransactReadOutcome::NotInstalled)); + return Ok(PartitionTransactReadQueryOutput( + PartitionTransactReadOutcome::NotInstalled, + )); }; if spec.table.id != input.table_id || spec.epoch != input.epoch { - return Ok(Json(PartitionTransactReadOutcome::StaleRoute)); + return Ok(PartitionTransactReadQueryOutput( + PartitionTransactReadOutcome::StaleRoute, + )); } match super::query_access(context)? { AccessState::Serving => {} - AccessState::Sealed => return Ok(Json(PartitionTransactReadOutcome::Sealed)), - AccessState::Importing => return Ok(Json(PartitionTransactReadOutcome::NotReady)), + AccessState::Sealed => { + return Ok(PartitionTransactReadQueryOutput( + PartitionTransactReadOutcome::Sealed, + )); + } + AccessState::Importing => { + return Ok(PartitionTransactReadQueryOutput( + PartitionTransactReadOutcome::NotReady, + )); + } } if input.operations.is_empty() || input.operations.len() > 100 { - return Ok(Json(PartitionTransactReadOutcome::Rejected { - index: 0, - reason: TransactionFailure::Validation( - "transaction operation count is outside 1..=100".into(), - ), - })); + return Ok(PartitionTransactReadQueryOutput( + PartitionTransactReadOutcome::Rejected { + index: 0, + reason: TransactionFailure::Validation( + "transaction operation count is outside 1..=100".into(), + ), + }, + )); } let images = match query_stage_operations(context, &spec, input.operations)? { Ok(images) => images, Err(error) => { - return Ok(Json(match error { + return Ok(PartitionTransactReadQueryOutput(match error { StageError::StaleRoute => PartitionTransactReadOutcome::StaleRoute, StageError::WrongPartition => PartitionTransactReadOutcome::WrongPartition, StageError::Rejected { index, reason } => { @@ -244,7 +293,9 @@ impl Query for PartitionTransactReadQuery { })); } }; - Ok(Json(PartitionTransactReadOutcome::Applied(images))) + Ok(PartitionTransactReadQueryOutput( + PartitionTransactReadOutcome::Applied(images), + )) } } @@ -534,6 +585,8 @@ fn apply_staged( mod participant; pub use participant::*; +mod prepare_batch; +pub use prepare_batch::*; pub(super) fn key_locked(context: &mut CommandContext<'_, '_>, key: &[u8]) -> Result { Ok(!context.sql(&lock_query(key, false))?[0].rows.is_empty()) diff --git a/src/partition/transaction/participant.rs b/src/partition/transaction/participant.rs index 14c59d1..711264c 100644 --- a/src/partition/transaction/participant.rs +++ b/src/partition/transaction/participant.rs @@ -52,88 +52,115 @@ impl Command for PreparePartitionTransaction { Json(input): Self::Input, ) -> Result> { let input = crate::transaction_transport::consume::(context, input)?; - let digest = blake3::hash(&serde_json::to_vec(&input)?); - if let Some(outcome) = crate::participant::prepared( - context, - input.transaction_id, - input.coordinator_cell, - digest, - )? { - return Ok(prepare_rejected(outcome)); - } - let Some(spec) = super::super::indexes::command_spec(context)? else { - return Ok(prepare_rejected(PrepareTransactionOutcome::NotInstalled)); - }; - if spec.table.id != input.table_id || spec.epoch != input.epoch { - return Ok(prepare_rejected(PrepareTransactionOutcome::StaleRoute)); - } - match command_access(context)? { - AccessState::Serving => {} - AccessState::Sealed => { - return Ok(prepare_rejected(PrepareTransactionOutcome::Sealed)); - } - AccessState::Importing => { - return Ok(prepare_rejected(PrepareTransactionOutcome::NotReady)); - } + prepare_partition(context, input) + } +} + +/// Prepare an inline participant with a bounded reply reservation. +/// A durable WideRequired rejection permits retry through the wide command. +pub struct PreparePartitionTransactionBounded; + +impl Command for PreparePartitionTransactionBounded { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 28; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + let result = prepare_partition(context, input)?; + crate::participant::bound_prepare_result(result) + } +} + +pub(super) fn prepare_partition( + context: &mut CommandContext<'_, '_>, + input: PreparePartitionTransactionInput, +) -> Result>> { + let digest = blake3::hash(&serde_json::to_vec(&input)?); + if let Some(outcome) = crate::participant::prepared( + context, + input.transaction_id, + input.coordinator_cell, + digest, + )? { + return Ok(prepare_rejected(outcome)); + } + let Some(spec) = super::super::indexes::command_spec(context)? else { + return Ok(prepare_rejected(PrepareTransactionOutcome::NotInstalled)); + }; + if spec.table.id != input.table_id || spec.epoch != input.epoch { + return Ok(prepare_rejected(PrepareTransactionOutcome::StaleRoute)); + } + match command_access(context)? { + AccessState::Serving => {} + AccessState::Sealed => { + return Ok(prepare_rejected(PrepareTransactionOutcome::Sealed)); } - if input.operations.is_empty() || input.operations.len() > 100 { - return Ok(prepare_validation( - 0, - "transaction operation count is outside 1..=100", - )); + AccessState::Importing => { + return Ok(prepare_rejected(PrepareTransactionOutcome::NotReady)); } - let staged = match stage_operations(context, &spec, input.operations)? { - Ok(staged) => staged, - Err(reason) => return Ok(prepare_rejected(reason.prepare_outcome())), - }; - let prepared = PreparedPartition { - table_id: input.table_id, - epoch: input.epoch, - images: staged, - }; - crate::participant::record_prepare( - context, - input.transaction_id, - input.coordinator_cell, - digest, - crate::participant::PreparedPayload { - bytes: serde_json::to_vec(&prepared)?, - operations: prepared.images.len(), - index_edits: prepared - .images - .iter() - .map(|image| image.index_capacity.edits) - .sum(), - index_overflow_bytes: prepared - .images - .iter() - .map(|image| image.index_capacity.overflow_bytes) - .sum(), - }, - &input.coordinator_key, - prepared + } + if input.operations.is_empty() || input.operations.len() > 100 { + return Ok(prepare_validation( + 0, + "transaction operation count is outside 1..=100", + )); + } + let staged = match stage_operations(context, &spec, input.operations)? { + Ok(staged) => staged, + Err(reason) => return Ok(prepare_rejected(reason.prepare_outcome())), + }; + let prepared = PreparedPartition { + table_id: input.table_id, + epoch: input.epoch, + images: staged, + }; + crate::participant::record_prepare( + context, + input.transaction_id, + input.coordinator_cell, + digest, + crate::participant::PreparedPayload { + bytes: serde_json::to_vec(&prepared)?, + operations: prepared.images.len(), + index_edits: prepared .images .iter() - .filter(|image| image.effect == StagedEffect::Read) - .map(|image| image.image.as_ref()), - )?; - for image in prepared.images { - context.sql(&statement( - "INSERT OR IGNORE INTO ddb_partition_transaction_locks \ - (item_key, partition_key, sort_key, transaction_id, write_lock) VALUES (?1, ?2, ?3, ?4, ?5)", - vec![ - SqlValue::Blob(image.key), - SqlValue::Blob(image.partition_key), - SqlValue::Blob(image.sort_key), - SqlValue::Blob(input.transaction_id.to_vec()), - SqlValue::Integer(i64::from(image.effect != StagedEffect::Read)), - ], - ))?; - } - Ok(CommandResult::Success(Json( - PrepareTransactionOutcome::Prepared, - ))) + .map(|image| image.index_capacity.edits) + .sum(), + index_overflow_bytes: prepared + .images + .iter() + .map(|image| image.index_capacity.overflow_bytes) + .sum(), + }, + &input.coordinator_key, + prepared + .images + .iter() + .filter(|image| image.effect == StagedEffect::Read) + .map(|image| image.image.as_ref()), + )?; + for image in prepared.images { + context.sql(&statement( + "INSERT OR IGNORE INTO ddb_partition_transaction_locks \ + (item_key, partition_key, sort_key, transaction_id, write_lock) VALUES (?1, ?2, ?3, ?4, ?5)", + vec![ + SqlValue::Blob(image.key), + SqlValue::Blob(image.partition_key), + SqlValue::Blob(image.sort_key), + SqlValue::Blob(input.transaction_id.to_vec()), + SqlValue::Integer(i64::from(image.effect != StagedEffect::Read)), + ], + ))?; } + Ok(CommandResult::Success(Json( + PrepareTransactionOutcome::Prepared, + ))) } fn prepare_rejected( @@ -229,3 +256,16 @@ impl Query for ReadPartitionTransactionResult { crate::participant::read_result(context, input) } } + +/// Read a saved partition image without reserving the wide item reply budget. +pub struct ReadPartitionTransactionResultBounded; +impl Query for ReadPartitionTransactionResultBounded { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 20; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = crate::BoundedTransactionReadResult; + fn execute(context: &mut QueryContext<'_>, Json(input): Self::Input) -> Result { + crate::participant::bounded_read_result(context, input) + } +} diff --git a/src/partition/transaction/prepare_batch.rs b/src/partition/transaction/prepare_batch.rs new file mode 100644 index 0000000..f2fc1d0 --- /dev/null +++ b/src/partition/transaction/prepare_batch.rs @@ -0,0 +1,66 @@ +//! Independent participant prepares sharing a durable publication. + +use std::collections::HashSet; + +use cellule_runtime::registry::{Command, CommandContext, CommandResult}; +use serde::{Deserialize, Serialize}; + +use super::participant::{PreparePartitionTransactionInput, prepare_partition}; +use crate::{Json, PrepareTransactionOutcome, Result}; + +pub(crate) const MAX_PREPARES: usize = 16; +pub(crate) const PREPARE_BATCH_BYTES: usize = 32 * 1024; + +/// Every input was prepared, or the complete application savepoint rolled back. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub enum PreparePartitionBatchOutcome { + /// All inputs have durable intents and locks at the returned sequence. + Prepared, + /// Use individual commands to preserve each rejection or replay outcome. + IndividualRequired, +} + +/// Coalesce bounded prepares without merging transaction identities or decisions. +pub struct PreparePartitionTransactionBatch; + +impl Command for PreparePartitionTransactionBatch { + const MODULE: &'static str = crate::DATA_MODULE; + const ID: u32 = 29; + const CODEC_VERSION: u32 = 1; + type Input = Json>; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(inputs): Self::Input, + ) -> Result> { + let mut identities = HashSet::new(); + if inputs.is_empty() + || inputs.len() > MAX_PREPARES + || serde_json::to_vec(&inputs)?.len() > PREPARE_BATCH_BYTES + || inputs + .iter() + .any(|input| !identities.insert(input.transaction_id)) + { + return Ok(individual_required()); + } + for input in inputs { + if !matches!( + prepare_partition(context, input)?, + CommandResult::Success(Json(PrepareTransactionOutcome::Prepared)) + ) { + // Rejection rolls back *all* intents, read images, reservations + // and locks created by this command. No individual failure + // image is truncated to fit a shared result envelope. + return Ok(individual_required()); + } + } + Ok(CommandResult::Success(Json( + PreparePartitionBatchOutcome::Prepared, + ))) + } +} + +fn individual_required() -> CommandResult> { + CommandResult::Rejected(Json(PreparePartitionBatchOutcome::IndividualRequired)) +} diff --git a/src/partition/update_batch.rs b/src/partition/update_batch.rs new file mode 100644 index 0000000..081d609 --- /dev/null +++ b/src/partition/update_batch.rs @@ -0,0 +1,260 @@ +//! Independent updates sharing one commit and retaining individual results. + +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; + +use super::*; + +pub(crate) const MAX_UPDATES: usize = 16; +pub(crate) const BATCH_INPUT_BYTES: u32 = 1024 * 1024; +pub(crate) const BATCH_OUTPUT_BYTES: u32 = 128 * 1024; + +pub(crate) const fn batch_operation(id: u32) -> OperationDescriptor { + OperationDescriptor { + input_limit: BATCH_INPUT_BYTES, + output_limit: BATCH_OUTPUT_BYTES, + ..operation(id) + } +} + +/// A routed update and the successful images required by its caller. +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct BatchedPartitionUpdate { + /// The same validation and expression contract as a standalone update. + pub input: PartitionUpdateInput, + /// Keep the previous successful image; condition failures always retain it. + pub return_old: bool, + /// Keep the committed new image. + pub return_new: bool, +} + +/// A compact reply, or a confirmed rollback requiring separate commands. +#[derive(Clone, Debug, PartialEq)] +pub enum PartitionUpdateBatchOutcome { + /// Every entry corresponds to its input, including condition failures. + Results(Vec), + /// All application writes were rolled back before this rejection committed. + IndividualRequired, +} + +impl WireValue for PartitionUpdateBatchOutcome { + fn encode(&self, encoder: &mut BoundedEncoder) -> std::result::Result<(), CodecError> { + let Self::Results(results) = self else { + return encoder.write_u8(0); + }; + if results.is_empty() || results.len() > MAX_UPDATES { + return Err(CodecError::Invalid("update batch count is outside 1..=16")); + } + encoder.write_u8(1)?; + encoder.write_count(results.len())?; + for result in results { + match result { + PartitionUpdateOutcome::Applied { old, new } => { + encoder.write_u8(0)?; + encoder.write_count(2)?; + crate::item_wire::encode_image(old.as_ref(), encoder)?; + crate::item_wire::encode_image(Some(new), encoder)?; + } + PartitionUpdateOutcome::ConditionFailed(old) => { + encoder.write_u8(1)?; + crate::item_wire::encode_images(std::slice::from_ref(old), encoder)?; + } + other => { + encoder.write_u8(2)?; + Json(other.clone()).encode(encoder)?; + } + } + } + Ok(()) + } + + fn decode(decoder: &mut BoundedDecoder<'_>) -> std::result::Result { + match decoder.read_u8()? { + 0 => return Ok(Self::IndividualRequired), + 1 => {} + _ => return Err(CodecError::Invalid("unknown update batch envelope")), + } + let count = decoder.read_count()?; + if count == 0 || count > MAX_UPDATES { + return Err(CodecError::Invalid("update batch count is outside 1..=16")); + } + let mut results = Vec::with_capacity(count); + for _ in 0..count { + results.push(match decoder.read_u8()? { + 0 => { + let mut images = crate::item_wire::decode_images(decoder)?; + if images.len() != 2 { + return Err(CodecError::Invalid("update result needs two images")); + } + let Some(Some(new)) = images.pop() else { + return Err(CodecError::Invalid("update result lacks new image")); + }; + PartitionUpdateOutcome::Applied { + old: images.pop().flatten(), + new, + } + } + 1 => { + let mut images = crate::item_wire::decode_images(decoder)?; + if images.len() != 1 { + return Err(CodecError::Invalid("condition result needs one image")); + } + PartitionUpdateOutcome::ConditionFailed(images.pop().flatten()) + } + 2 => { + let Json(result) = Json::::decode(decoder)?; + if matches!( + result, + PartitionUpdateOutcome::Applied { .. } + | PartitionUpdateOutcome::ConditionFailed(_) + ) { + return Err(CodecError::Invalid("noncanonical update result envelope")); + } + result + } + _ => return Err(CodecError::Invalid("unknown update result tag")), + }); + } + Ok(Self::Results(results)) + } +} + +/// Coalesce distinct item updates, isolating ordinary validation rejections. +pub struct PartitionUpdateBatch; + +impl Command for PartitionUpdateBatch { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 25; + const CODEC_VERSION: u32 = 1; + type Input = Json>; + type Output = PartitionUpdateBatchOutcome; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(inputs): Self::Input, + ) -> Result> { + execute(context, inputs, BATCH_OUTPUT_BYTES) + } +} + +/// One update with a larger compact reply after confirmed batch rollback. +pub struct PartitionUpdateIndividual; + +impl Command for PartitionUpdateIndividual { + const MODULE: &'static str = DATA_MODULE; + const ID: u32 = 26; + const CODEC_VERSION: u32 = 1; + type Input = Json>; + type Output = PartitionUpdateBatchOutcome; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(inputs): Self::Input, + ) -> Result> { + if inputs.len() != 1 { + return Err(Error::Command("individual update requires one operation")); + } + execute(context, inputs, OPERATION_BYTES) + } +} + +fn execute( + context: &mut CommandContext<'_, '_>, + inputs: Vec, + limit: u32, +) -> Result> { + if inputs.is_empty() || inputs.len() > MAX_UPDATES { + return Err(Error::Command("update batch count is outside 1..=16")); + } + for (index, input) in inputs.iter().enumerate() { + if inputs[..index].iter().any(|previous| { + previous.input.table_id == input.input.table_id && previous.input.key == input.input.key + }) { + return Err(Error::Command("update batch repeats an item key")); + } + } + let mut results = Vec::with_capacity(inputs.len()); + for (ordinal, update) in inputs.into_iter().enumerate() { + // Every ordinary rejection in execute_partition_update precedes the + // first application write. A storage/stream error instead propagates + // and rolls back the entire command; it is never treated as isolated. + let result = super::execute_partition_update(context, update.input, true, ordinal)?; + let Json(mut result) = match result { + CommandResult::Success(result) | CommandResult::Rejected(result) => result, + }; + if let PartitionUpdateOutcome::Applied { old, new } = &mut result { + if !update.return_old { + *old = None; + } + if !update.return_new { + new.clear(); + } + } + results.push(result); + } + let output = PartitionUpdateBatchOutcome::Results(results); + let mut encoder = BoundedEncoder::new(limit)?; + match output.encode(&mut encoder) { + Ok(()) => Ok(CommandResult::Success(output)), + Err(CodecError::Limit) if limit == BATCH_OUTPUT_BYTES => { + // Cellule's application savepoint rolls back items, indexes, + // capacity reservations and stream records before recording this + // rejected receipt. Only that durable rejection permits fallback. + Ok(CommandResult::Rejected( + PartitionUpdateBatchOutcome::IndividualRequired, + )) + } + Err(error) => Err(error.into()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn decode(bytes: &[u8]) -> std::result::Result { + let mut decoder = BoundedDecoder::new(bytes, OPERATION_BYTES)?; + let result = PartitionUpdateBatchOutcome::decode(&mut decoder)?; + decoder.finish()?; + Ok(result) + } + + #[test] + fn update_batch_codec_rejects_unbounded_counts_truncation_and_trailing_bytes() { + for bytes in [ + vec![1, 0xff, 0xff, 0xff, 0xff], + vec![1, 0, 0, 0, 0], + vec![2], + vec![0, 0], + ] { + assert!(decode(&bytes).is_err()); + } + let item = Item::from([( + "padding".into(), + extenddb_core::types::AttributeValue::S("\0".repeat(390_000)), + )]); + let expected = + PartitionUpdateBatchOutcome::Results(vec![PartitionUpdateOutcome::Applied { + old: Some(item.clone()), + new: item, + }]); + let mut encoder = BoundedEncoder::new(OPERATION_BYTES).unwrap(); + expected.encode(&mut encoder).unwrap(); + let bytes = encoder.finish(); + assert!( + bytes.len() < 800_000, + "escaped images must retain compact size" + ); + assert_eq!(decode(&bytes).unwrap(), expected); + for length in [0, 1, 4, 8, bytes.len() - 1] { + assert!(decode(&bytes[..length]).is_err()); + } + let mut trailing = bytes; + trailing.push(0); + assert!(decode(&trailing).is_err()); + let mut small = BoundedEncoder::new(BATCH_OUTPUT_BYTES).unwrap(); + assert!(matches!( + expected.encode(&mut small), + Err(CodecError::Limit) + )); + } +} diff --git a/src/provision.rs b/src/provision.rs index 05c58d9..5365012 100644 --- a/src/provision.rs +++ b/src/provision.rs @@ -1,6 +1,8 @@ //! Initial table data Cell admission through the existing Cell runtime. mod capacity; +mod coordinator_bootstrap; +mod coordinator_registration; mod directory; mod global_indexes; mod ranges; @@ -62,7 +64,9 @@ pub struct CellInitialPartitionProvisioner { directory: PathBuf, initial_partition_count: u16, transaction_recovery: transactions::CoordinatorRecovery, - admission: tokio::sync::Mutex<()>, + admission: tokio::sync::RwLock<()>, + coordinator_bootstrap: coordinator_bootstrap::CoordinatorBootstrap, + coordinator_registrations: coordinator_registration::CoordinatorRegistrations, peers: Option>, recovery_transport: Option>, } @@ -92,6 +96,8 @@ impl CellInitialPartitionProvisioner { initial_partition_count: 1, transaction_recovery: Default::default(), admission: Default::default(), + coordinator_bootstrap: Default::default(), + coordinator_registrations: Default::default(), peers: None, recovery_transport: None, }) @@ -185,7 +191,7 @@ impl CellInitialPartitionProvisioner { .try_into() .map_err(|_| StorageError::Internal("invalid coordinator partition".into()))?; let client = CellClient::local(self.application.registry(), account_handle); - client + let registration = client .command::( &account, mutation_identity()?, @@ -196,6 +202,8 @@ impl CellInitialPartitionProvisioner { ) .await .map_err(cell_error)?; + self.remember_coordinator_registration(&account, &target, registration.receipt.incarnation) + .await?; Ok(handle) } @@ -540,8 +548,21 @@ impl CellInitialPartitionProvisioner { nodes: &NodeDirectory, ) -> Result { let target = credential_target(access_key_id).map_err(provision_error)?; - let proof = self.cataloged(&target, crate::credentials::MODULE).await?; - self.takeover_expired(&target, proof, nodes, initialize_credentials) + self.takeover_expired_credential_cell(&target, nodes).await + } + + pub(crate) async fn takeover_expired_credential_cell( + &self, + target: &CellTarget, + nodes: &NodeDirectory, + ) -> Result { + if target.namespace() != crate::credentials::NAMESPACE { + return Err(StorageError::Validation( + "invalid credential namespace".into(), + )); + } + let proof = self.cataloged(target, crate::credentials::MODULE).await?; + self.takeover_expired(target, proof, nodes, initialize_credentials) .await } @@ -635,7 +656,7 @@ impl CellInitialPartitionProvisioner { nodes: &NodeDirectory, initialize: for<'a> fn(&rusqlite::Transaction<'a>) -> cellule_runtime::Result<()>, ) -> Result { - let _admission = self.admission.lock().await; + let _admission = self.admission.write().await; self.reclaim_settled_capacity(target).await?; let authority = CellAuthority::new(self.layout.clone()); let mut observed = authority @@ -689,7 +710,10 @@ impl CellInitialPartitionProvisioner { MAX_NODE_LOG_RECOVERY_CELLS, ) .await - .map_err(provision_error)?; + .map_err(|error| { + tracing::warn!(?error, former = ?former_session, claimant = ?self.session, "node-log recovery failed"); + provision_error(error) + })?; // Recovery pins the overlay by changing Cell authority. // The takeover must see that new control and its scratch. observed = authority @@ -812,18 +836,26 @@ impl CellInitialPartitionProvisioner { module: &'static str, initialize: for<'a> fn(&rusqlite::Transaction<'a>) -> cellule_runtime::Result<()>, ) -> Result { + let proof = self.provision_module_catalog(target, module).await?; + self.admit_initialized(target, proof, initialize) + .await + .map_err(provision_error) + } + + async fn provision_module_catalog( + &self, + target: &CellTarget, + module: &'static str, + ) -> Result { let code = self .application .registry() .module_code(module) .ok_or_else(|| StorageError::Internal("Cell module is not compiled".into()))?; - let proof = CellCatalog::new(self.layout.clone(), target.tenant()) + CellCatalog::new(self.layout.clone(), target.tenant()) .provision( CatalogEntry::new(target, CatalogRole::Sql, code, 1).map_err(provision_error)?, ) - .await - .map_err(provision_error)?; - self.admit_initialized(target, proof, initialize) .await .map_err(provision_error) } @@ -844,9 +876,20 @@ impl CellInitialPartitionProvisioner { proof: CatalogProof, initialize: for<'a> fn(&rusqlite::Transaction<'a>) -> cellule_runtime::Result<()>, ) -> cellule_runtime::Result { + if let Some(handle) = self + .try_bootstrap_coordinator( + target, + proof.clone(), + initialize, + coordinator_bootstrap::AuthorityObservation::Read, + ) + .await? + { + return Ok(handle); + } // Serialize local activation/reclamation. Authority CAS still decides // ownership against other nodes; this guard never fences peers. - let _admission = self.admission.lock().await; + let _admission = self.admission.write().await; let authority = CellAuthority::new(self.layout.clone()); let observed = authority.load(target.cell_id()).await?; if let Some(observed) = &observed @@ -895,6 +938,18 @@ impl CellInitialPartitionProvisioner { { return self.activate_published(target, proof, observed).await; } + self.bootstrap_unpublished(target, proof, observed, initialize) + .await + } + + async fn bootstrap_unpublished( + &self, + target: &CellTarget, + proof: CatalogProof, + observed: cellule_runtime::control::authority::VersionedControl, + initialize: for<'a> fn(&rusqlite::Transaction<'a>) -> cellule_runtime::Result<()>, + ) -> cellule_runtime::Result { + let authority = CellAuthority::new(self.layout.clone()); let replica = CellReplica::new( self.layout.clone(), *target.cell_id().as_bytes(), @@ -1044,7 +1099,7 @@ impl InitialPartitionProvisioner for CellInitialPartitionProvisioner { ) -> BoxedFuture<'a, Result, StorageError>> { Box::pin(async move { let account = account_target(account_id).map_err(provision_error)?; - let crate::TablePlacement::Routed { initial_partitions } = table.placement else { + let Some(initial_partitions) = table.placement.initial_partitions() else { return Err(StorageError::Validation( "account-local table has no initial ranges".into(), )); @@ -1093,7 +1148,7 @@ impl InitialPartitionProvisioner for CellInitialPartitionProvisioner { ) -> BoxedFuture<'a, Result, StorageError>> { Box::pin(async move { let account = account_target(account_id).map_err(provision_error)?; - let crate::TablePlacement::Routed { initial_partitions } = table.placement else { + let Some(initial_partitions) = table.placement.initial_partitions() else { return Err(StorageError::Validation( "account-local table has no initial ranges".into(), )); diff --git a/src/provision/capacity.rs b/src/provision/capacity.rs index 483811f..6a24c31 100644 --- a/src/provision/capacity.rs +++ b/src/provision/capacity.rs @@ -206,6 +206,9 @@ impl CellInitialPartitionProvisioner { let Some(partition) = partition else { return Ok(false); }; + if index_record.is_none() && table.placement == crate::TablePlacement::Single { + return Ok(false); + } if let Some(index) = index_record { self.split_global_index_if_over_database_bytes( account_id, @@ -432,6 +435,15 @@ impl CellInitialPartitionProvisioner { if usage.database_bytes <= max_database_bytes { return Ok(None); } + let state = client + .query::(&target, None, Json(())) + .await + .map_err(cell_error)? + .output + .0; + if state.is_some_and(|state| state.spec.table.placement == crate::TablePlacement::Single) { + return Ok(None); + } self.split_partition( account_id, client.clone(), @@ -511,6 +523,11 @@ impl CellInitialPartitionProvisioner { "split source contract changed".into(), )); } + if source.table.placement == crate::TablePlacement::Single { + return Err(StorageError::Validation( + "single-Cell tables cannot split their base data Cell".into(), + )); + } let directory = self .split_ready_directory(&client, account_id, table_id, lower) .await?; diff --git a/src/provision/coordinator_bootstrap.rs b/src/provision/coordinator_bootstrap.rs new file mode 100644 index 0000000..8497d29 --- /dev/null +++ b/src/provision/coordinator_bootstrap.rs @@ -0,0 +1,202 @@ +//! Bounded concurrent initialization of fresh coordinator Cells. + +use std::sync::{ + Mutex, + atomic::{AtomicU64, Ordering}, +}; + +use cellule_runtime::cell::actor::{CellHandle, CellRuntime}; +use cellule_runtime::control::{ControlState, Owner, authority::CellAuthority}; +use cellule_runtime::identity::{CellTarget, IncarnationId}; +use cellule_runtime::{Error, Result}; + +use super::CellInitialPartitionProvisioner; + +const GATES: usize = 64; + +pub(super) enum AuthorityObservation { + Read, + Missing(u64), +} + +pub(super) struct CoordinatorBootstrap { + gates: [tokio::sync::Mutex<()>; GATES], + pending: Mutex, + generations: [AtomicU64; GATES], +} + +impl Default for CoordinatorBootstrap { + fn default() -> Self { + Self { + gates: std::array::from_fn(|_| tokio::sync::Mutex::new(())), + pending: Mutex::new(0), + generations: std::array::from_fn(|_| AtomicU64::new(0)), + } + } +} + +impl CoordinatorBootstrap { + pub(super) fn generation(&self, target: &CellTarget) -> u64 { + self.generations[stripe(target)].load(Ordering::Acquire) + } + + fn reserve(&self, runtime: &CellRuntime) -> Result>> { + let mut pending = self.pending.lock().map_err(|_| Error::RuntimeClosed)?; + let stats = runtime.stats(); + // All exclusive product admission/reclamation waits for our read guards. + // Cellule's active count includes its in-flight activation reservations. + // Counting those again while this guard exists is conservative. Reading + // stats under this mutex prevents a completion from disappearing between + // the active and pending observations. + if stats.active_cells().saturating_add(*pending) >= stats.active_cell_capacity() { + return Ok(None); + } + *pending += 1; + Ok(Some(PendingBootstrap { + pending: &self.pending, + })) + } +} + +struct PendingBootstrap<'a> { + pending: &'a Mutex, +} + +struct Creating<'a>(&'a AtomicU64); + +impl<'a> Creating<'a> { + fn new(generation: &'a AtomicU64) -> Self { + generation.fetch_add(1, Ordering::Release); + Self(generation) + } +} + +impl Drop for Creating<'_> { + fn drop(&mut self) { + // Also invalidate absences observed while creation was in flight. + // Cancellation must advance the generation before releasing the stripe. + self.0.fetch_add(1, Ordering::Release); + } +} + +impl Drop for PendingBootstrap<'_> { + fn drop(&mut self) { + let mut pending = match self.pending.lock() { + Ok(pending) => pending, + Err(poisoned) => poisoned.into_inner(), + }; + *pending = pending.saturating_sub(1); + } +} + +impl CellInitialPartitionProvisioner { + pub(super) async fn admit_missing_coordinator( + &self, + target: &CellTarget, + generation: u64, + proof: cellule_runtime::cell::catalog::CatalogProof, + ) -> std::result::Result { + if let Some(handle) = self + .try_bootstrap_coordinator( + target, + proof.clone(), + super::initialize_coordinator, + AuthorityObservation::Missing(generation), + ) + .await + .map_err(super::provision_error)? + { + return Ok(handle); + } + // A creation race or exhausted capacity needs the normal path, which + // reloads canonical authority before restoring or reclaiming anything. + self.admit_initialized(target, proof, super::initialize_coordinator) + .await + .map_err(super::provision_error) + } + + pub(super) async fn try_bootstrap_coordinator( + &self, + target: &CellTarget, + proof: cellule_runtime::cell::catalog::CatalogProof, + initialize: for<'a> fn(&cellule_ltx::rusqlite::Transaction<'a>) -> Result<()>, + observation: AuthorityObservation, + ) -> Result> { + if target.namespace() != crate::transaction_coordinator::NAMESPACE { + return Ok(None); + } + // Fixed stripes bound memory and serialize identical Cells. Unrelated + // stripes can initialize concurrently; collisions only delay admission. + let stripe = stripe(target); + let _cell = self.coordinator_bootstrap.gates[stripe].lock().await; + let _admission = self.admission.read().await; + let authority = CellAuthority::new(self.layout.clone()); + let observed = match observation { + AuthorityObservation::Missing(generation) + if self.coordinator_bootstrap.generations[stripe].load(Ordering::Acquire) + == generation => + { + None + } + // Only reuse absence. The create-if-absent CAS below revalidates it; + // an observed owner or root must always be loaded afresh. + // Waiting behind a local creation invalidates the earlier absence. + // Re-read before trying a CAS that the store would retry on conflict. + AuthorityObservation::Read | AuthorityObservation::Missing(_) => { + authority.load(target.cell_id()).await? + } + }; + if let Some(observed) = &observed { + if let Some(handle) = self.runtime.local_handle(proof.clone(), observed).await? { + self.track_coordinator(target) + .map_err(super::admission_error)?; + return Ok(Some(handle)); + } + if observed.value().root.is_some() + || observed.value().state != ControlState::Recovering + || observed.value().owner.as_ref().map(|owner| owner.session) != Some(self.session) + { + // Published restore and peer ownership retain exclusive admission. + return Ok(None); + } + } + let Some(_reservation) = self.coordinator_bootstrap.reserve(&self.runtime)? else { + // Drop the shared guards before the caller reclaims under exclusivity. + return Ok(None); + }; + let observed = match observed { + Some(observed) => observed, + None => { + // Entry and exit invalidate concurrent absence observations. + // Colliding stripes only cause an extra canonical GET. + let _creating = Creating::new(&self.coordinator_bootstrap.generations[stripe]); + let incarnation = IncarnationId::from_bytes(*uuid::Uuid::now_v7().as_bytes()); + match authority + .create_initial( + &proof, + incarnation, + Owner { + session: self.session, + endpoint: self.endpoint.clone(), + }, + ) + .await + { + Ok(observed) => observed, + Err(Error::CellAlreadyActive) => return Ok(None), + Err(error) => return Err(error), + } + } + }; + // Before runtime admission, cancellation drops the local reservation. + // After admission, Cellule retains its own counted activation reservation + // until completion/cleanup, even when this request stops awaiting it. + self.bootstrap_unpublished(target, proof, observed, initialize) + .await + .map(Some) + } +} + +fn stripe(target: &CellTarget) -> usize { + usize::from(target.cell_id().as_bytes()[0]) % GATES +} diff --git a/src/provision/coordinator_registration.rs b/src/provision/coordinator_registration.rs new file mode 100644 index 0000000..0093f15 --- /dev/null +++ b/src/provision/coordinator_registration.rs @@ -0,0 +1,131 @@ +//! Coalesce account discovery updates; BEGIN still waits for durable registration. + +use std::collections::{HashMap, VecDeque}; +use std::sync::{Arc, Weak}; +use std::time::Duration; + +use cellule_runtime::client::CellClient; +use cellule_runtime::identity::{CellId, CellTarget, IncarnationId}; +use extenddb_storage::error::StorageError; +use tokio::sync::{Mutex, OwnedSemaphorePermit, Semaphore, oneshot}; + +use crate::backend::{cell_error, mutation_identity}; +use crate::transaction_coordinator::MAX_REGISTRATION_SHARDS; +use crate::{Json, RegisterCoordinatorShards, RegisterCoordinatorShardsInput}; + +const WINDOW: Duration = Duration::from_millis(2); + +#[derive(Default)] +pub(super) struct CoordinatorRegistrations { + slots: Mutex>>, +} + +struct Slot { + account: CellTarget, + account_id: String, + client: CellClient, + capacity: Arc, + queue: Mutex, +} + +#[derive(Default)] +struct Queue { + pending: VecDeque, + running: bool, +} + +struct Pending { + shard: u32, + reply: oneshot::Sender>, + _permit: OwnedSemaphorePermit, +} + +impl CoordinatorRegistrations { + pub(super) async fn register( + &self, + client: &CellClient, + account: &CellTarget, + account_id: &str, + shard: u32, + ) -> Result { + let slot = { + let mut slots = self.slots.lock().await; + if let Some(slot) = slots.get(&account.cell_id()).and_then(Weak::upgrade) { + slot + } else { + slots.retain(|_, slot| slot.strong_count() > 0); + let slot = Arc::new(Slot { + account: account.clone(), + account_id: account_id.into(), + client: client.clone(), + capacity: Arc::new(Semaphore::new(MAX_REGISTRATION_SHARDS * 4)), + queue: Mutex::new(Queue::default()), + }); + slots.insert(account.cell_id(), Arc::downgrade(&slot)); + slot + } + }; + let permit = slot.capacity.clone().acquire_owned().await.map_err(|_| { + StorageError::Transient("coordinator registration queue stopped".into()) + })?; + let (reply, result) = oneshot::channel(); + let start = { + let mut queue = slot.queue.lock().await; + queue.pending.push_back(Pending { + shard, + reply, + _permit: permit, + }); + let start = !queue.running; + queue.running = true; + start + }; + if start { + tokio::spawn(flush(slot)); + } + // Cancellation after enqueue does not discard a discovery update. + // It still cannot permit BEGIN: the caller needs this durable receipt. + result.await.map_err(|_| { + StorageError::Transient("coordinator registration worker stopped".into()) + })? + } +} + +async fn flush(slot: Arc) { + loop { + if slot.queue.lock().await.pending.len() < MAX_REGISTRATION_SHARDS { + tokio::time::sleep(WINDOW).await; + } + let batch: Vec<_> = { + let mut queue = slot.queue.lock().await; + if queue.pending.is_empty() { + queue.running = false; + return; + } + let count = queue.pending.len().min(MAX_REGISTRATION_SHARDS); + queue.pending.drain(..count).collect() + }; + let mut shards: Vec<_> = batch.iter().map(|pending| pending.shard).collect(); + shards.sort_unstable(); + shards.dedup(); + let outcome = match mutation_identity() { + Ok(identity) => slot + .client + .command::( + &slot.account, + identity, + Json(RegisterCoordinatorShardsInput { + account_id: slot.account_id.clone(), + shards, + }), + ) + .await + .map(|acknowledged| acknowledged.receipt.incarnation) + .map_err(cell_error), + Err(error) => Err(error), + }; + for pending in batch { + let _ = pending.reply.send(outcome.clone()); + } + } +} diff --git a/src/provision/directory.rs b/src/provision/directory.rs index f422a8a..e635be8 100644 --- a/src/provision/directory.rs +++ b/src/provision/directory.rs @@ -310,26 +310,16 @@ impl CellInitialPartitionProvisioner { .await .map_err(cell_error)?; } - if !self - .retire_directory_step(client, account_id, &spec) + let Some(sequence) = self + .retire_directory_receipt(client, account_id, &spec) .await? - { + else { return Ok(false); - } - let directory = directory_target(account_id, &spec).map_err(provision_error)?; - let observed = client - .query::(&directory, None, Json(())) - .await - .map_err(cell_error)?; - if !observed - .output - .0 - .is_some_and(|state| state.spec == spec && state.mode == DirectoryMode::Retired) - { - return Err(StorageError::Transient( - "directory retirement is not complete".into(), - )); - } + }; + // The traversal verified this immutable root's terminal state and + // observed its durable sequence. Residency may be released before + // the account acknowledgement; rereading the root is unnecessary + // and can fail for a local maintenance client after reclamation. client .command::( &account, @@ -337,7 +327,7 @@ impl CellInitialPartitionProvisioner { Json(crate::TableDirectoryRetirement { table_id: table_id.into(), directory_id: spec.table_id, - sequence: observed.receipt.commit_sequence, + sequence, }), ) .await @@ -365,6 +355,19 @@ impl CellInitialPartitionProvisioner { account_id: &str, root: &DirectorySpec, ) -> Result { + self.retire_directory_receipt(client, account_id, root) + .await + .map(|receipt| receipt.is_some()) + } + + // Some(sequence) proves the exact root and every planned descendant are + // durably retired. None retains the next bounded child step for recovery. + async fn retire_directory_receipt( + &self, + client: &CellClient, + account_id: &str, + root: &DirectorySpec, + ) -> Result, StorageError> { let mut current = root.clone(); let mut parent = None; let mut published = true; @@ -423,7 +426,7 @@ impl CellInitialPartitionProvisioner { match mode { DirectoryMode::Retired => { let Some(parent) = parent else { - return Ok(true); + return Ok(Some(sequence)); }; let parent_client = self .existing_directory_client(client, account_id, &parent) @@ -441,7 +444,8 @@ impl CellInitialPartitionProvisioner { ) .await .map_err(cell_error)?; - return Ok(parent == *root && recorded.output.0); + return Ok((parent == *root && recorded.output.0) + .then_some(recorded.receipt.commit_sequence)); } DirectoryMode::Retiring { children, diff --git a/src/provision/rebalance.rs b/src/provision/rebalance.rs index e1d8c49..c3b681e 100644 --- a/src/provision/rebalance.rs +++ b/src/provision/rebalance.rs @@ -178,7 +178,7 @@ impl CellInitialPartitionProvisioner { { // Serialize release with local restore and reclamation. The actor // rechecks generation and settled work against foreground races. - let _admission = self.admission.lock().await; + let _admission = self.admission.write().await; if let Err(error) = self .runtime .release_idle_cell(intent.cell, self.session, intent.generation) diff --git a/src/provision/residency.rs b/src/provision/residency.rs index 85406bc..bb01277 100644 --- a/src/provision/residency.rs +++ b/src/provision/residency.rs @@ -164,7 +164,7 @@ impl CellInitialPartitionProvisioner { if self.runtime.stats().active_cells() < self.runtime.stats().active_cell_capacity() { return Ok(()); } - let _admission = self.admission.lock().await; + let _admission = self.admission.write().await; // Placement samples the local pool before activation. Apply the same // reclamation policy here so a full pool cannot hide a restorable root. self.reclaim_settled_capacity(target).await @@ -250,7 +250,7 @@ impl CellInitialPartitionProvisioner { { return Ok(None); } - let _admission = self.admission.lock().await; + let _admission = self.admission.write().await; let authority = CellAuthority::new(self.layout.clone()); let observed = authority .load(target.cell_id()) @@ -356,7 +356,7 @@ impl CellInitialPartitionProvisioner { // Gather local state before metadata reads can restore owners and // release ranges. Admission excludes competing product reclamation; // remote discovery below must run without this recursive gate. - let _admission = self.admission.lock().await; + let _admission = self.admission.write().await; let stats = self.runtime.stats(); if stats.active_cells() < stats.active_cell_capacity() { return Ok(()); @@ -532,7 +532,7 @@ impl CellInitialPartitionProvisioner { }; // Metadata lookup may itself restore an idle account. Hold the local // admission gate only for release, avoiding recursive admission waits. - let _admission = self.admission.lock().await; + let _admission = self.admission.write().await; let stats = self.runtime.stats(); if stats.active_cells() < stats.active_cell_capacity() { return Ok(()); diff --git a/src/provision/transactions.rs b/src/provision/transactions.rs index 257a6f0..9963655 100644 --- a/src/provision/transactions.rs +++ b/src/provision/transactions.rs @@ -8,9 +8,10 @@ use std::{ }; use cellule_host::CellNodeTaskGroup; +use cellule_runtime::cell::catalog::CatalogRole; use cellule_runtime::client::CellClient; use cellule_runtime::control::{ControlState, authority::CellAuthority}; -use cellule_runtime::identity::CellTarget; +use cellule_runtime::identity::{CellTarget, IncarnationId}; use cellule_runtime::node::NodeDirectory; use cellule_runtime::partition_for_shard; use extenddb_storage::error::StorageError; @@ -34,6 +35,17 @@ type Initialize = #[derive(Default)] pub(super) struct CoordinatorRecovery { shards: RwLock>, + registrations: RwLock>, +} + +// Registrations are monotonic within an account incarnation. This receipt only +// removes repeated admission reads; actor dispatch retains its normal fences. +const MAX_RESIDENT_REGISTRATIONS: usize = 4_096; + +#[derive(Clone, Copy, PartialEq, Eq)] +struct ResidentRegistration { + account: IncarnationId, + coordinator: IncarnationId, } #[derive(Clone)] @@ -44,6 +56,97 @@ struct RecoveryShard { } impl CellInitialPartitionProvisioner { + async fn resident_registration( + &self, + account: &CellTarget, + coordinator: &CellTarget, + ) -> Result, StorageError> { + let (account_handle, coordinator_handle) = tokio::try_join!( + self.runtime.resident_handle(account, CatalogRole::Sql), + self.runtime.resident_handle(coordinator, CatalogRole::Sql), + ) + .map_err(provision_error)?; + let (Some(account_handle), Some(coordinator_handle)) = (account_handle, coordinator_handle) + else { + return Ok(None); + }; + let registry = self.application.registry(); + if Some(account_handle.code()) != registry.module_code(crate::MODULE) + || Some(coordinator_handle.code()) + != registry.module_code(crate::transaction_coordinator::MODULE) + || account_handle.schema() != 1 + || coordinator_handle.schema() != 1 + { + return Ok(None); + } + Ok(Some(ResidentRegistration { + account: account_handle.incarnation(), + coordinator: coordinator_handle.incarnation(), + })) + } + + async fn has_resident_registration( + &self, + account: &CellTarget, + coordinator: &CellTarget, + ) -> Result { + let cached = self + .transaction_recovery + .registrations + .read() + .map_err(|_| StorageError::Internal("coordinator registration cache poisoned".into()))? + .get(coordinator.cell_id().as_bytes()) + .copied(); + let Some(cached) = cached else { + return Ok(false); + }; + if self.resident_registration(account, coordinator).await? == Some(cached) { + return Ok(true); + } + let mut registrations = self + .transaction_recovery + .registrations + .write() + .map_err(|_| { + StorageError::Internal("coordinator registration cache poisoned".into()) + })?; + // A concurrent admission may have recorded a newer incarnation. + if registrations.get(coordinator.cell_id().as_bytes()) == Some(&cached) { + registrations.remove(coordinator.cell_id().as_bytes()); + } + Ok(false) + } + + pub(super) async fn remember_coordinator_registration( + &self, + account: &CellTarget, + coordinator: &CellTarget, + account_incarnation: IncarnationId, + ) -> Result<(), StorageError> { + let Some(resident) = self.resident_registration(account, coordinator).await? else { + return Ok(()); + }; + if resident.account != account_incarnation { + return Ok(()); + } + let mut registrations = self + .transaction_recovery + .registrations + .write() + .map_err(|_| { + StorageError::Internal("coordinator registration cache poisoned".into()) + })?; + let cell = *coordinator.cell_id().as_bytes(); + if registrations.len() >= MAX_RESIDENT_REGISTRATIONS + && !registrations.contains_key(&cell) + && let Some(evicted) = registrations.keys().next().copied() + { + registrations.remove(&evicted); + } + registrations.insert(cell, resident); + Ok(()) + } + pub(super) fn track_coordinator(&self, target: &CellTarget) -> Result<(), StorageError> { if target.namespace() != crate::transaction_coordinator::NAMESPACE { return Ok(()); @@ -208,6 +311,11 @@ impl CellInitialPartitionProvisioner { after: &mut Option, ) -> Result<(), StorageError> { let account = account_target(account_id).map_err(provision_error)?; + // The index can be owned by the same failed node as the unknown shard. + // Restore Idle/expired account authority before routing its discovery + // query; a live remote owner remains in place under the normal fence. + self.recover_discovered_owner(&account, crate::MODULE, initialize_account, nodes) + .await?; let page = client .query::( &account, @@ -675,6 +783,9 @@ impl crate::CoordinatorProvisioner for CellInitialPartitionProvisioner { let account = account_target(account_id).map_err(provision_error)?; let target = crate::coordinator_target(account_id, routing_key).map_err(provision_error)?; + if self.has_resident_registration(&account, &target).await? { + return Ok(()); + } let shard = u32::from_be_bytes( target .partition() @@ -687,23 +798,21 @@ impl crate::CoordinatorProvisioner for CellInitialPartitionProvisioner { }; // Registration is discovery, not residency. A released shard must // restore its published root before token lookup or a new BEGIN. - // These reads use independent stores: the registration is in the - // account Cell and the authority record is local node metadata. - // Start them together so admission pays for the slower read once. - let (registered, observed) = tokio::try_join!( + // Registration, authority and immutable catalog publication are + // independent. Start them together; bootstrap still requires the + // published catalog proof, and BEGIN still waits for registration. + let missing_generation = self.coordinator_bootstrap.generation(&target); + let ((registered, registration_receipt), observed, proof) = tokio::try_join!( async { - Ok::<_, StorageError>( - client - .query::( - &account, - None, - Json(input.clone()), - ) - .await - .map_err(cell_error)? - .output - .0, - ) + let registration = client + .query::( + &account, + None, + Json(input.clone()), + ) + .await + .map_err(cell_error)?; + Ok::<_, StorageError>((registration.output.0, registration.receipt)) }, async { CellAuthority::new(self.layout.clone()) @@ -711,6 +820,7 @@ impl crate::CoordinatorProvisioner for CellInitialPartitionProvisioner { .await .map_err(provision_error) }, + self.provision_module_catalog(&target, crate::transaction_coordinator::MODULE), )?; if registered && observed @@ -724,10 +834,9 @@ impl crate::CoordinatorProvisioner for CellInitialPartitionProvisioner { if observed.as_ref().is_some_and(|record| { record.value().owner.is_some() && record.value().root.is_some() }) { - // Another node may have published this shard before registration. + // Catalog provision validates the exact immutable entry even + // when another node published this shard before registration. // Keep its authority; the routed client will reach that owner. - self.cataloged(&target, crate::transaction_coordinator::MODULE) - .await?; if observed .as_ref() .and_then(|record| record.value().owner.as_ref()) @@ -737,24 +846,51 @@ impl crate::CoordinatorProvisioner for CellInitialPartitionProvisioner { } } else { self.reclaim_retired_ranges(client, &account, None).await?; - self.admit_module( - &target, - crate::transaction_coordinator::MODULE, - initialize_coordinator, - ) - .await?; + if observed.is_none() { + self.admit_missing_coordinator(&target, missing_generation, proof) + .await?; + } else { + self.admit_initialized(&target, proof, initialize_coordinator) + .await + .map_err(provision_error)?; + } } if registered { + // A query can observe a registration while publication is still + // in flight. Cache it only when the account's published root + // covers that observation; otherwise use normal admission again. + if self + .resident_registration(&account, &target) + .await? + .is_none() + { + return Ok(()); + } + let published = CellAuthority::new(self.layout.clone()) + .load(account.cell_id()) + .await + .map_err(provision_error)?; + if published.as_ref().is_some_and(|control| { + control.value().incarnation == registration_receipt.incarnation + && control.value().root.as_ref().is_some_and(|root| { + root.commit_sequence >= registration_receipt.commit_sequence + }) + }) { + self.remember_coordinator_registration( + &account, + &target, + registration_receipt.incarnation, + ) + .await?; + } return Ok(()); } - client - .command::( - &account, - crate::backend::mutation_identity()?, - Json(input), - ) - .await - .map_err(cell_error)?; + let account_incarnation = self + .coordinator_registrations + .register(client, &account, account_id, shard) + .await?; + self.remember_coordinator_registration(&account, &target, account_incarnation) + .await?; Ok(()) }) } diff --git a/src/server.rs b/src/server.rs index 2fbc4e6..cb93d9c 100644 --- a/src/server.rs +++ b/src/server.rs @@ -9,6 +9,7 @@ mod node_log_recovery; mod node_log_sender; mod peer_receiver; mod placement; +mod runtime_metrics; pub use capacity::measured_node_capacity; pub use node_lease::{NodeLeasePublisher, PublishedNodeLease}; @@ -16,6 +17,7 @@ pub use node_log_authority::PublishedNodeLogAuthority; pub use node_log_provider::PeerNodeDurabilityProvider; pub use node_log_recovery::recover_fenced_node_log; pub use node_log_sender::PeerNodeLogTransport; +pub use runtime_metrics::RuntimeMetrics; use std::sync::Arc; @@ -106,6 +108,7 @@ pub struct BeyonddbPeers { registry: Arc, placement: Arc, follower: Option, + follower_metrics: Option>, } impl BeyonddbPeers { @@ -147,6 +150,7 @@ impl BeyonddbPeers { round_trip, }), follower: None, + follower_metrics: None, }) } @@ -160,13 +164,27 @@ impl BeyonddbPeers { store: Arc, guard: cellule_runtime::NodeLeaseGuard, ) -> Self { - self.follower = Some(node_log_receiver::FollowerEndpoint::new( + let mut endpoint = node_log_receiver::FollowerEndpoint::new( self.placement.directory.clone(), self.runtime.clone(), node, store, guard, - )); + ); + if let Some(metrics) = &self.follower_metrics { + endpoint = endpoint.with_runtime_metrics(metrics.clone()); + } + self.follower = Some(endpoint); + self + } + + /// Observe authenticated follower append phases on the private listener. + #[must_use] + pub fn with_follower_metrics(mut self, metrics: Arc) -> Self { + if let Some(endpoint) = self.follower.take() { + self.follower = Some(endpoint.with_runtime_metrics(metrics.clone())); + } + self.follower_metrics = Some(metrics); self } @@ -177,9 +195,9 @@ impl BeyonddbPeers { /// Build a client with the opt-in short-lived local owner cache. /// - /// The cache keeps resident handles for 500 ms while the Cell handle still - /// fences drained owners. Authority is re-read after expiry, so ownership - /// changes remain bounded by the cache window. + /// The cache keeps resident handles for 500 ms, then resolves the current + /// actor without provider reads. Reuse checks the current resident owner; + /// dispatch fences drained handles. Remote routing keeps exact checks. pub fn client_with_cache( &self, provisioner: Arc, @@ -205,7 +223,19 @@ impl BeyonddbPeers { /// Build the authenticated peer route; mount only on this identity's mTLS listener. pub fn router(&self, provisioner: Arc) -> axum::Router { - let router = peer_receiver::peer_router(self, provisioner); + self.router_with_cache(provisioner, false) + } + + /// Build the authenticated peer route with the opt-in 500 ms active owner handle cache. + /// + /// Cell handles still fence drained owners; peer enrollment and request + /// authorization are checked for every invocation. + pub fn router_with_cache( + &self, + provisioner: Arc, + handle_cache_enabled: bool, + ) -> axum::Router { + let router = peer_receiver::peer_router(self, provisioner, handle_cache_enabled); match &self.follower { Some(follower) => router.merge(node_log_receiver::router(follower.clone())), None => router, diff --git a/src/server/node_lease.rs b/src/server/node_lease.rs index cde9d92..d2b0338 100644 --- a/src/server/node_lease.rs +++ b/src/server/node_lease.rs @@ -2,12 +2,12 @@ use std::{ sync::Arc, - time::{Duration, SystemTime, UNIX_EPOCH}, + time::{Duration, Instant, SystemTime, UNIX_EPOCH}, }; use cellule_runtime::node::{NodeAdvertisement, NodeDirectory, VersionedNodeAdvertisement}; use cellule_runtime::{Error, NodeLeaseGuard, Result}; -use tokio::sync::Mutex; +use tokio::sync::{Mutex, watch}; use tokio_util::sync::CancellationToken; // Use the runtime's maximum advertisement lifetime for storage refresh headroom. @@ -19,6 +19,50 @@ const FENCE_MARGIN: Duration = Duration::from_secs(1); type SignAdvertisement = dyn Fn(i64, i64) -> Result + Send + Sync; +#[derive(Clone, Copy)] +struct RenewalProgress { + phase: &'static str, + started: Instant, +} + +fn renewal_phase(progress: &watch::Sender, phase: &'static str) { + progress.send_replace(RenewalProgress { + phase, + started: Instant::now(), + }); +} + +/// Latest completed canonical observation of this publisher's own boot session. +/// +/// This is an ETag hint for a conditional write, never a liveness or authority +/// cache. No provider I/O holds the watch lock. Late responses cannot replace a +/// newer generation; unseen changes still require the directory's CAS rebase. +#[derive(Clone)] +pub(super) struct SessionObservation(watch::Sender); + +impl SessionObservation { + fn new(observed: VersionedNodeAdvertisement) -> Self { + Self(watch::channel(observed).0) + } + + fn latest(&self) -> VersionedNodeAdvertisement { + self.0.borrow().clone() + } + + pub(super) fn record(&self, observed: &VersionedNodeAdvertisement) { + self.0.send_if_modified(|current| { + if observed.advertisement().session() == current.advertisement().session() + && observed.advertisement().generation() > current.advertisement().generation() + { + current.clone_from(observed); + true + } else { + false + } + }); + } +} + /// Publishes signed node advertisements into the authoritative object-store directory. pub struct NodeLeasePublisher { directory: NodeDirectory, @@ -44,7 +88,7 @@ impl NodeLeasePublisher { /// Publish the initial lease before installing it in a Cell node. pub async fn publish(self) -> Result { let now_ms = unix_time_ms()?; - let advertisement = self.advertisement(now_ms).await?; + let advertisement = self.advertisement(now_ms, None).await?; let observed = self.directory.create(advertisement, now_ms).await?; // Object-store publication can take time; lease the remaining // authoritative window, not a fresh window after the response. @@ -52,23 +96,34 @@ impl NodeLeasePublisher { Ok(PublishedNodeLease { publisher: self, session: observed.advertisement().session(), - observed: Arc::new(Mutex::new(observed)), + observed: SessionObservation::new(observed), + log_transitions: Arc::new(Mutex::new(())), guard, fence_on_drop: true, }) } - async fn advertisement(&self, now_ms: i64) -> Result { + async fn advertisement( + &self, + now_ms: i64, + progress: Option<&watch::Sender>, + ) -> Result { let expires = lease_expiry(now_ms)?; let sign = Arc::clone(&self.sign); + let progress = progress.cloned(); // Only one sample is in flight per publisher. Its timestamp precedes // dispatch, so queue/probe latency cannot extend the signed lease. - tokio::task::spawn_blocking(move || sign(now_ms, expires)) - .await - .map_err(|source| Error::Facility { - name: "node-capacity-signing", - source: Box::new(source), - })? + tokio::task::spawn_blocking(move || { + if let Some(progress) = &progress { + renewal_phase(progress, "capacity-signing"); + } + sign(now_ms, expires) + }) + .await + .map_err(|source| Error::Facility { + name: "node-capacity-signing", + source: Box::new(source), + })? } } @@ -76,7 +131,8 @@ impl NodeLeasePublisher { pub struct PublishedNodeLease { publisher: NodeLeasePublisher, session: cellule_runtime::identity::SessionId, - observed: Arc>, + observed: SessionObservation, + log_transitions: Arc>, guard: NodeLeaseGuard, fence_on_drop: bool, } @@ -98,7 +154,8 @@ impl PublishedNodeLease { self.publisher.directory.clone(), self.session, self.guard.clone(), - Arc::clone(&self.observed), + Arc::clone(&self.log_transitions), + self.observed.clone(), ) } @@ -109,6 +166,10 @@ impl PublishedNodeLease { /// serving composition retires the session through `shutdown_serving_node`. pub async fn run(mut self, cancellation: &CancellationToken) -> Result<()> { let guard = self.guard.clone(); + let (progress, observed_progress) = watch::channel(RenewalProgress { + phase: "heartbeat-timer", + started: Instant::now(), + }); // A storage request can outlive the lease or shutdown. Dropping its // future cannot revoke a remote CAS, but must never renew this guard. tokio::select! { @@ -116,51 +177,77 @@ impl PublishedNodeLease { self.fence_on_drop = false; Ok(()) }, - () = guard.wait_fenced() => Err(Error::Fenced), - result = self.renew() => result, + () = guard.wait_fenced() => { + let observed = *observed_progress.borrow(); + tracing::warn!( + diagnostic = "node-lease-renewal-phase", + session = ?self.session, + phase = observed.phase, + phase_elapsed_ms = observed.started.elapsed().as_secs_f64() * 1000.0, + "serving node lease fenced while waiting for renewal", + ); + Err(Error::Fenced) + }, + result = self.renew(&progress) => result, } } - async fn renew(&mut self) -> Result<()> { + async fn renew(&mut self, progress: &watch::Sender) -> Result<()> { let mut ticks = tokio::time::interval(HEARTBEAT); ticks.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); ticks.tick().await; loop { + renewal_phase(progress, "heartbeat-timer"); ticks.tick().await; loop { - match self.refresh().await { + match self.refresh(progress).await { Ok(()) => break, - Err(error) if self.guard.remaining() > FENCE_MARGIN => { - tokio::time::sleep(RETRY).await; - self.guard.check()?; - if self.guard.remaining() <= FENCE_MARGIN { + Err(error) => { + let observed = *progress.borrow(); + tracing::warn!( + diagnostic = "node-lease-renewal-phase", + session = ?self.session, + phase = observed.phase, + phase_elapsed_ms = observed.started.elapsed().as_secs_f64() * 1000.0, + lease_remaining_ms = self.guard.remaining().as_secs_f64() * 1000.0, + error = %error, + "serving node lease refresh failed", + ); + if self.guard.remaining() > FENCE_MARGIN { + renewal_phase(progress, "retry-delay"); + tokio::time::sleep(RETRY).await; + self.guard.check()?; + if self.guard.remaining() <= FENCE_MARGIN { + self.guard.fence(); + return Err(error); + } + } else { self.guard.fence(); return Err(error); } } - Err(error) => { - self.guard.fence(); - return Err(error); - } } } } } - async fn refresh(&mut self) -> Result<()> { + async fn refresh(&mut self, progress: &watch::Sender) -> Result<()> { self.guard.check()?; let now_ms = unix_time_ms()?; - let next = self.publisher.advertisement(now_ms).await?; - let mut observed = self.observed.lock().await; + renewal_phase(progress, "capacity-queue"); + let next = self.publisher.advertisement(now_ms, Some(progress)).await?; self.guard.check()?; + renewal_phase(progress, "directory-refresh"); + let observed = self.observed.latest(); let renewed = self .publisher .directory .refresh(&observed, next, now_ms) .await?; + renewal_phase(progress, "guard-renewal"); self.guard .renew(unix_time_ms()?, renewed.advertisement().expires_at_ms())?; - *observed = renewed; + self.observed.record(&renewed); Ok(()) } } diff --git a/src/server/node_log_authority.rs b/src/server/node_log_authority.rs index ec490f5..24635ce 100644 --- a/src/server/node_log_authority.rs +++ b/src/server/node_log_authority.rs @@ -16,21 +16,23 @@ use cellule_runtime::{ }, }; -use super::node_lease::unix_time_ms; +use super::node_lease::{SessionObservation, unix_time_ms}; const MAX_CAS_ATTEMPTS: usize = 4; /// Directory-backed authority for one leased node session. /// -/// Each operation reloads the current version so a concurrent heartbeat does -/// not overwrite or permanently obstruct an enrolled log transition. The -/// directory still validates every transition and compares the exact ETag. +/// Each operation reloads the current version and retries CAS races. Log I/O +/// never holds the heartbeat's local state: a stalled transition must not +/// prevent renewal. The directory validates every transition and exact ETag. +/// Log transitions share a separate mutex; heartbeat renewal never takes it. #[derive(Clone)] pub struct PublishedNodeLogAuthority { directory: NodeDirectory, session: SessionId, guard: NodeLeaseGuard, - observed: Arc>, + transitions: Arc>, + observed: SessionObservation, } impl PublishedNodeLogAuthority { @@ -42,17 +44,19 @@ impl PublishedNodeLogAuthority { directory: NodeDirectory, session: SessionId, guard: NodeLeaseGuard, - observed: Arc>, + transitions: Arc>, + observed: SessionObservation, ) -> Self { Self { directory, session, guard, + transitions, observed, } } - async fn load(&self, observed: &mut VersionedNodeAdvertisement) -> Result { + async fn load(&self) -> Result<(VersionedNodeAdvertisement, i64)> { self.guard.check()?; let now_ms = unix_time_ms()?; let current = self @@ -61,8 +65,8 @@ impl PublishedNodeLogAuthority { .await? .ok_or(Error::Fenced)?; self.guard.check()?; - *observed = current; - Ok(now_ms) + self.observed.record(¤t); + Ok((current, now_ms)) } /// Enroll a complete follower set for the next log epoch, if available. @@ -75,10 +79,10 @@ impl PublishedNodeLogAuthority { required_follower_bytes: u64, live_node_limit: usize, ) -> Result>> { - let mut observed = self.observed.lock().await; + let _transition = self.transitions.lock().await; let mut last_error = None; for _ in 0..MAX_CAS_ATTEMPTS { - let now_ms = self.load(&mut observed).await?; + let (observed, now_ms) = self.load().await?; if let Some(log) = observed.advertisement().log() { if log.epoch() != log_epoch || log.phase() != NodeLogPhase::Open || log.active() { return Err(Error::Node("node session has a different or active log")); @@ -97,9 +101,9 @@ impl PublishedNodeLogAuthority { .await { Ok(Some(enrolled)) => { - *observed = enrolled; self.guard.check()?; - let log = observed + self.observed.record(&enrolled); + let log = enrolled .advertisement() .log() .ok_or(Error::Node("enrolled node log is missing"))?; @@ -115,19 +119,49 @@ impl PublishedNodeLogAuthority { )) } + /// Checks the exact current epoch's members before host-owned rotation. + /// + /// Read the canonical enrollment independently of heartbeat state. + /// A slow membership scan must not prevent this node from renewing its + /// own lease. The result requests rotation; it grants no append authority. + pub async fn rotation_required(&self, log_epoch: u64, live_node_limit: usize) -> Result { + self.guard.check()?; + let current = self + .directory + .load(self.session, unix_time_ms()?) + .await? + .ok_or(Error::Fenced)?; + self.guard.check()?; + self.observed.record(¤t); + let members = exact_open_log(¤t, log_epoch)?.members().to_vec(); + let live = self + .directory + .live(unix_time_ms()?, live_node_limit) + .await?; + self.guard.check()?; + // Scan I/O can outlive a follower advertisement. Recheck expiry at + // completion rather than treating its pre-scan timestamp as fresh. + let now_ms = unix_time_ms()?; + Ok(!members.iter().all(|member| { + live.iter().any(|advertisement| { + advertisement.node() == *member && advertisement.expires_at_ms() > now_ms + }) + })) + } + async fn activate_epoch(&self, log_epoch: u64) -> Result<()> { - let mut observed = self.observed.lock().await; + let _transition = self.transitions.lock().await; let mut last_error = None; for _ in 0..MAX_CAS_ATTEMPTS { - let now_ms = self.load(&mut observed).await?; + let (observed, now_ms) = self.load().await?; let log = exact_open_log(&observed, log_epoch)?; if log.active() { return Ok(()); } match self.directory.activate_log(&observed, now_ms).await { Ok(updated) => { - *observed = updated; self.guard.check()?; + self.observed.record(&updated); return Ok(()); } Err(error) => last_error = Some(error), @@ -140,10 +174,10 @@ impl PublishedNodeLogAuthority { } async fn advance_epoch_coverage(&self, log_epoch: u64, tiered_through: u64) -> Result<()> { - let mut observed = self.observed.lock().await; + let _transition = self.transitions.lock().await; let mut last_error = None; for _ in 0..MAX_CAS_ATTEMPTS { - let now_ms = self.load(&mut observed).await?; + let (observed, now_ms) = self.load().await?; let log = exact_open_log(&observed, log_epoch)?; if log.tiered_through() >= tiered_through { return Ok(()); @@ -154,8 +188,8 @@ impl PublishedNodeLogAuthority { .await { Ok(updated) => { - *observed = updated; self.guard.check()?; + self.observed.record(&updated); return Ok(()); } Err(error) => last_error = Some(error), @@ -168,16 +202,16 @@ impl PublishedNodeLogAuthority { } async fn close_epoch(&self, barrier: &NodeLogRotationBarrier) -> Result<()> { + let _transition = self.transitions.lock().await; if barrier.leader_session() != self.session { return Err(Error::Node( "node-log close barrier belongs to another session", )); } - let mut observed = self.observed.lock().await; let mut last_error = None; let mut attempted = false; for _ in 0..MAX_CAS_ATTEMPTS { - let now_ms = self.load(&mut observed).await?; + let (observed, now_ms) = self.load().await?; let Some(log) = observed.advertisement().log() else { return if attempted { Ok(()) @@ -191,8 +225,8 @@ impl PublishedNodeLogAuthority { attempted = true; match self.directory.close_log(&observed, barrier, now_ms).await { Ok(updated) => { - *observed = updated; self.guard.check()?; + self.observed.record(&updated); return Ok(()); } Err(error) => last_error = Some(error), diff --git a/src/server/node_log_provider.rs b/src/server/node_log_provider.rs index 2beba1e..823baa9 100644 --- a/src/server/node_log_provider.rs +++ b/src/server/node_log_provider.rs @@ -5,7 +5,7 @@ use std::{ pin::Pin, sync::{ Arc, - atomic::{AtomicU64, Ordering}, + atomic::{AtomicBool, AtomicU64, Ordering}, }, }; @@ -26,7 +26,7 @@ use super::{PeerNodeLogTransport, PublishedNodeLogAuthority}; /// Recruits authoritative follower sets for the host's durability supervisor. /// /// Constructing this adapter does not install it or enable follower proofs. -/// The serving binary must complete owner recovery before installing it. +/// A serving host must gate recruitment until owner recovery completes. pub struct PeerNodeDurabilityProvider { authority: Arc, transport: Arc, @@ -35,6 +35,7 @@ pub struct PeerNodeDurabilityProvider { guard: NodeLeaseGuard, telemetry: CellTelemetryHandle, next_epoch: AtomicU64, + recruitment_ready: Arc, } impl PeerNodeDurabilityProvider { @@ -64,8 +65,16 @@ impl PeerNodeDurabilityProvider { guard, telemetry, next_epoch: AtomicU64::new(1), + recruitment_ready: Arc::new(AtomicBool::new(true)), }) } + + /// Install during host startup, but defer recruitment until recovery is ready. + #[must_use] + pub fn with_recruitment_gate(mut self, ready: Arc) -> Self { + self.recruitment_ready = ready; + self + } } impl NodeDurabilityProvider for PeerNodeDurabilityProvider { @@ -77,15 +86,27 @@ impl NodeDurabilityProvider for PeerNodeDurabilityProvider { ) -> Pin>> + Send>> { Box::pin(async move { self.guard.check()?; + if !self.recruitment_ready.load(Ordering::Acquire) { + return Ok(None); + } let epoch = self.next_epoch.load(Ordering::Acquire); let Some(members) = self .authority .recruit(epoch, required_follower_bytes, live_node_limit) - .await? + .await + .inspect_err(|error| { + tracing::warn!( + log_epoch = epoch, + error = %error, + "node-log follower enrollment failed" + ); + })? else { + tracing::debug!(log_epoch = epoch, "node-log follower ensemble unavailable"); return Ok(None); }; self.guard.check()?; + tracing::debug!(log_epoch = epoch, ?members, "node-log followers enrolled"); let transport: Arc = self.transport.clone(); let authority: Arc = self.authority.clone(); Ok(Some(NodeDurabilityConfig::new( @@ -102,7 +123,26 @@ impl NodeDurabilityProvider for PeerNodeDurabilityProvider { }) } + fn rotation_required( + self: Arc, + live_node_limit: usize, + ) -> Pin> + Send>> { + Box::pin(async move { + // The host serializes recruitment, installation and rotation. + // A failed/incomplete epoch must not be reported as healthy by + // consulting a cached or newly selected candidate member set. + Ok(self + .authority + .rotation_required(self.next_epoch.load(Ordering::Acquire), live_node_limit) + .await?) + }) + } + fn rotation_event(&self, event: NodeDurabilityRotation) { + tracing::debug!(?event, "node-log durability rotation"); + if event == NodeDurabilityRotation::Failed { + tracing::warn!("node-log durability supervisor step failed"); + } if event == NodeDurabilityRotation::Started { self.next_epoch.fetch_add(1, Ordering::AcqRel); } diff --git a/src/server/node_log_receiver.rs b/src/server/node_log_receiver.rs index 70a7fbc..4cee964 100644 --- a/src/server/node_log_receiver.rs +++ b/src/server/node_log_receiver.rs @@ -21,7 +21,10 @@ use cellule_runtime::{ use futures_util::StreamExt; use tokio::sync::Semaphore; -use super::node_lease::unix_time_ms; +use super::{ + node_lease::unix_time_ms, + runtime_metrics::{FollowerPhase, RuntimeMetrics, observe_follower}, +}; pub(super) const MEDIA_TYPE: &str = "application/vnd.beyonddb.node-log-v1"; const MAGIC: &[u8; 4] = b"BNL1"; @@ -42,6 +45,7 @@ pub(super) struct FollowerEndpoint { store: Arc, guard: NodeLeaseGuard, permits: Arc, + metrics: Option>, } impl FollowerEndpoint { @@ -59,9 +63,15 @@ impl FollowerEndpoint { store, guard, permits: Arc::new(Semaphore::new(MAX_CONCURRENT_REQUESTS)), + metrics: None, } } + pub(super) fn with_runtime_metrics(mut self, metrics: Arc) -> Self { + self.metrics = Some(metrics); + self + } + async fn dispatch(&self, request: WireRequest, identity: PeerTlsIdentity) -> Result> { self.dispatch_with_identity(request, identity.certificate(), identity.public_key()) .await @@ -75,9 +85,16 @@ impl FollowerEndpoint { ) -> Result> { self.guard.check()?; let now_ms = unix_time_ms()?; - self.directory - .peer_verifier(request.caller, certificate, public_key, now_ms) - .await?; + let append_metrics = matches!(&request.operation, Operation::Append(_)) + .then_some(self.metrics.as_deref()) + .flatten(); + let enrollment = observe_follower( + append_metrics, + FollowerPhase::Enrollment, + self.directory + .peer_verifier(request.caller, certificate, public_key, now_ms), + ) + .await?; self.guard.check()?; let now_ms = unix_time_ms()?; let response = match request.operation { @@ -87,20 +104,23 @@ impl FollowerEndpoint { "follower append caller is not leader", )); } - self.directory - .authorize_log_append( - request.leader, - self.member, - request.epoch, - request.argument, - now_ms, - ) - .await?; + // Use this request's fresh mTLS-bound canonical observation; + // recheck expiry after provider I/O before any durable append. + enrollment.authorize_log_append( + self.member, + request.epoch, + request.argument, + now_ms, + )?; self.guard.check()?; encode_receipt( - self.store - .append(request.leader, request.epoch, frames, request.argument) - .await?, + observe_follower( + append_metrics, + FollowerPhase::DurableAppend, + self.store + .append(request.leader, request.epoch, frames, request.argument), + ) + .await?, ) } Operation::Seal => { @@ -492,9 +512,13 @@ mod tests { ltx::{CellStorageLayout, DiskBudget, Host, Limits}, node::{NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeFailureDomain}, }; - use cellule_store::Store; + use cellule_store::{Store, test_support::CountingObjectStore}; use ed25519_dalek::SigningKey; - use object_store::{memory::InMemory, path::Path}; + use object_store::{ + memory::InMemory, + path::Path, + throttle::{ThrottleConfig, ThrottledStore}, + }; fn request(tag: u8, count: u32, frames: &[&[u8]]) -> Bytes { let mut encoded = Vec::new(); @@ -611,13 +635,20 @@ mod tests { #[tokio::test] async fn authenticated_append_survives_reopen_and_rejects_wrong_peer() { let limits = Limits::default(); + let counted = Arc::new(CountingObjectStore::new(Arc::new(ThrottledStore::new( + InMemory::new(), + ThrottleConfig { + wait_get_per_call: std::time::Duration::from_millis(5), + ..ThrottleConfig::default() + }, + )))); let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), + Store::new(counted.clone()), Path::from("follower-receiver-test"), [42; 16], ); let directory = NodeDirectory::new( - layout, + layout.clone(), Digest::from_bytes([80; 32]), Digest::from_bytes([81; 32]), Digest::from_bytes([82; 32]), @@ -702,13 +733,56 @@ mod tests { .await .is_err() ); + assert!( + endpoint + .dispatch_with_identity(append(), Digest::from_bytes([21; 32]), [99; 32]) + .await + .is_err() + ); + assert!( + endpoint + .dispatch_with_identity( + WireRequest { + caller: follower, + ..append() + }, + Digest::from_bytes([22; 32]), + SigningKey::from_bytes(&[22; 32]).verifying_key().to_bytes(), + ) + .await + .is_err() + ); + assert!( + endpoint + .dispatch_with_identity( + WireRequest { + argument: 1, + ..append() + }, + Digest::from_bytes([21; 32]), + leader_key, + ) + .await + .is_err() + ); assert_eq!(store.retained_bytes(), 0); for _ in 0..2 { + counted.reset(); let result = endpoint .dispatch_with_identity(append(), Digest::from_bytes([21; 32]), leader_key) .await .unwrap(); assert_eq!(&result[result.len() - 8..], &1_u64.to_be_bytes()); + let requests = counted.requests(); + assert_eq!( + requests.len(), + 1, + "each append needs one canonical enrollment read" + ); + assert_eq!( + requests[0].location, + layout.node_path(leader.as_bytes()).to_string() + ); } assert!( endpoint diff --git a/src/server/node_log_sender.rs b/src/server/node_log_sender.rs index 5ef76a8..3718816 100644 --- a/src/server/node_log_sender.rs +++ b/src/server/node_log_sender.rs @@ -2,7 +2,7 @@ use std::{ collections::VecDeque, - sync::{Arc, Mutex}, + sync::{Arc, Mutex, Weak}, time::Duration, }; @@ -23,10 +23,13 @@ use reqwest::{StatusCode, Url, header}; use super::{ node_lease::unix_time_ms, node_log_receiver::{self, Operation, WireRequest}, + runtime_metrics::{FollowerPhase, RuntimeMetrics, observe_follower}, }; const PEER_CACHE_MS: i64 = 500; const MAX_CACHED_PEERS: usize = 128; +// Match the reviewed NodeDirectory::resolve_node live-scan bound. +const MAX_DISCOVERY_NODES: usize = 1_024; const RESPONSE_TIMEOUT: Duration = Duration::from_secs(32); const MAX_TOTAL_TAIL_BYTES: usize = 64 * 1024 * 1024; const MAX_TAIL_PAGES: usize = 4096; @@ -44,6 +47,8 @@ pub struct PeerNodeLogTransport { guard: NodeLeaseGuard, local_store: Option>, peers: Arc>>, + discovery: Arc>>>, + metrics: Option>, } #[derive(Clone)] @@ -58,6 +63,11 @@ struct CachedPeer { client: reqwest::Client, } +struct FleetDiscovery { + observed_at_ms: i64, + advertisements: Vec, +} + impl PeerNodeLogTransport { /// Return the boot session whose lease and TLS identity bind this transport. pub const fn session(&self) -> SessionId { @@ -85,9 +95,18 @@ impl PeerNodeLogTransport { guard, local_store: None, peers: Arc::new(Mutex::new(VecDeque::new())), + discovery: Arc::new(Mutex::new(Weak::new())), + metrics: None, } } + /// Enable bounded append-phase observations, including cancelled requests. + #[must_use] + pub fn with_runtime_metrics(mut self, metrics: Arc) -> Self { + self.metrics = Some(metrics); + self + } + /// Allow a recovery claimant to seal/read its own persistent follower lane. /// /// Local recovery still requires the directory's fenced-owner claim. @@ -123,6 +142,9 @@ impl PeerNodeLogTransport { if member == self.node { return Err(Error::PeerAuthorization("follower is the local node")); } + if member.as_bytes().iter().all(|byte| *byte == 0) { + return Err(Error::Node("node identity is zero")); + } let now_ms = unix_time_ms()?; if let Some(peer) = self .peers @@ -138,13 +160,49 @@ impl PeerNodeLogTransport { { return Ok(peer); } - let advertisement = self - .directory - .resolve_node(member, now_ms) - .await? + // The shipper appends to its distinct members concurrently. Share that + // cohort's complete signed scan instead of scanning the fleet once per + // follower. Only in-flight lookups keep this observation alive; normal + // peer-cache freshness and hard advertisement expiry remain unchanged. + let discovery = { + let mut current = self + .discovery + .lock() + .map_err(|_| Error::Peer("node-log discovery state is poisoned"))?; + match current.upgrade() { + Some(discovery) => discovery, + None => { + let discovery = Arc::new(tokio::sync::OnceCell::new()); + *current = Arc::downgrade(&discovery); + discovery + } + } + }; + let observed = discovery + .get_or_try_init(|| async { + self.guard.check()?; + let observed_at_ms = unix_time_ms()?; + let advertisements = self + .directory + .live(observed_at_ms, MAX_DISCOVERY_NODES) + .await?; + self.guard.check()?; + Ok::<_, Error>(FleetDiscovery { + observed_at_ms, + advertisements, + }) + }) + .await?; + let advertisement = observed + .advertisements + .iter() + .find(|advertisement| { + advertisement.node() == member && advertisement.expires_at_ms() > now_ms + }) + .cloned() .ok_or(Error::Peer("follower has no live advertisement"))?; self.guard.check()?; - self.refresh_peer(member, advertisement, now_ms) + self.refresh_peer(member, advertisement, observed.observed_at_ms) } fn refresh_peer( @@ -196,10 +254,23 @@ impl PeerNodeLogTransport { } async fn send(&self, member: NodeId, request: WireRequest, tail: bool) -> Result { + let append_metrics = matches!(&request.operation, Operation::Append(_)) + .then_some(self.metrics.as_deref()) + .flatten(); let encoded = request .encode() .map_err(|()| Error::Peer("invalid node-log request"))?; - let peer = self.peer(member).await?; + let peer = + observe_follower(append_metrics, FollowerPhase::PeerLookup, self.peer(member)).await?; + observe_follower( + append_metrics, + FollowerPhase::RoundTrip, + self.send_resolved(peer, encoded, tail), + ) + .await + } + + async fn send_resolved(&self, peer: CachedPeer, encoded: Vec, tail: bool) -> Result { let url = peer .endpoint .join("internal/node-log/v1") @@ -314,6 +385,14 @@ impl NodeLogTransport for PeerNodeLogTransport { }, ) .await + .inspect_err(|error| { + tracing::warn!( + ?member, + log_epoch = request.log_epoch, + error = %error, + "node-log follower append failed" + ); + }) }) } @@ -483,7 +562,7 @@ mod tests { use std::{ path::{Path, PathBuf}, process::Command, - sync::Arc, + sync::{Arc, atomic::Ordering}, }; use cellule_ltx::{Db, NodeFrameScope, encode_node_frame}; @@ -501,6 +580,97 @@ mod tests { use cellule_store::Store; use object_store::{memory::InMemory, path::Path as ObjectPath}; + #[derive(Debug, Default)] + struct DelayedDiscoveryStore { + inner: InMemory, + counted_path: Mutex>, + reads: std::sync::atomic::AtomicUsize, + gate: Mutex>, + } + + #[derive(Debug)] + struct DiscoveryGate { + entered: Arc, + release: Arc, + } + + impl std::fmt::Display for DelayedDiscoveryStore { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("DelayedDiscoveryStore") + } + } + + #[async_trait::async_trait] + impl object_store::ObjectStore for DelayedDiscoveryStore { + async fn put_opts( + &self, + location: &ObjectPath, + payload: object_store::PutPayload, + options: object_store::PutOptions, + ) -> object_store::Result { + self.inner.put_opts(location, payload, options).await + } + + async fn put_multipart_opts( + &self, + location: &ObjectPath, + options: object_store::PutMultipartOptions, + ) -> object_store::Result> { + self.inner.put_multipart_opts(location, options).await + } + + async fn get_opts( + &self, + location: &ObjectPath, + options: object_store::GetOptions, + ) -> object_store::Result { + let counted = self.counted_path.lock().unwrap().as_ref() == Some(location); + if counted { + self.reads.fetch_add(1, Ordering::SeqCst); + let gate = self.gate.lock().unwrap().take(); + if let Some(gate) = gate { + gate.entered.add_permits(1); + gate.release.acquire().await.unwrap().forget(); + } + // Make the real lookup yield so all cold transport callers + // enter discovery before the first result can populate cache. + tokio::time::sleep(Duration::from_millis(40)).await; + } + self.inner.get_opts(location, options).await + } + + fn delete_stream( + &self, + locations: futures_util::stream::BoxStream<'static, object_store::Result>, + ) -> futures_util::stream::BoxStream<'static, object_store::Result> { + self.inner.delete_stream(locations) + } + + fn list( + &self, + prefix: Option<&ObjectPath>, + ) -> futures_util::stream::BoxStream<'static, object_store::Result> + { + self.inner.list(prefix) + } + + async fn list_with_delimiter( + &self, + prefix: Option<&ObjectPath>, + ) -> object_store::Result { + self.inner.list_with_delimiter(prefix).await + } + + async fn copy_opts( + &self, + from: &ObjectPath, + to: &ObjectPath, + options: object_store::CopyOptions, + ) -> object_store::Result<()> { + self.inner.copy_opts(from, to, options).await + } + } + fn run(command: &mut Command) { assert!(command.output().unwrap().status.success()); } @@ -602,6 +772,8 @@ mod tests { #[tokio::test] async fn pinned_mtls_append_survives_follower_store_reopen() { + let sender_metrics = Arc::new(RuntimeMetrics::default()); + let receiver_metrics = Arc::new(RuntimeMetrics::default()); let root = tempfile::TempDir::new().unwrap(); let ca_key = root.path().join("ca.key"); let ca = root.path().join("ca.crt"); @@ -626,18 +798,23 @@ mod tests { .arg(&ca)); let (leader_cert, leader_key) = certificate(root.path(), "leader", &ca, &ca_key); let (follower_cert, follower_key) = certificate(root.path(), "follower", &ca, &ca_key); + let (second_cert, second_key) = certificate(root.path(), "second-follower", &ca, &ca_key); let leader_tls = LoadedPeerTls::load(&leader_cert, &leader_key, &ca, "localhost").unwrap(); let follower_tls = LoadedPeerTls::load(&follower_cert, &follower_key, &ca, "localhost").unwrap(); + let second_tls = LoadedPeerTls::load(&second_cert, &second_key, &ca, "localhost").unwrap(); + let second_listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let second_endpoint = format!("https://{}", second_listener.local_addr().unwrap()); let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); let follower_endpoint = format!("https://{}", listener.local_addr().unwrap()); + let discovery_store = Arc::new(DelayedDiscoveryStore::default()); let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), + Store::new(discovery_store.clone()), ObjectPath::from("follower-transport-test"), [42; 16], ); let directory = NodeDirectory::new( - layout, + layout.clone(), leader_tls.fleet(), Digest::from_bytes([81; 32]), Digest::from_bytes([82; 32]), @@ -646,13 +823,15 @@ mod tests { let follower_session = SessionId::from_bytes([2; 16]); let leader_node = NodeId::from_bytes([3; 16]); let follower_node = NodeId::from_bytes([4; 16]); + let second_session = SessionId::from_bytes([7; 16]); + let second_node = NodeId::from_bytes([6; 16]); let now_ms = unix_time_ms().unwrap(); directory .create( advertisement( follower_node, follower_session, - follower_endpoint, + follower_endpoint.clone(), &follower_tls, true, true, @@ -663,6 +842,22 @@ mod tests { ) .await .unwrap(); + directory + .create( + advertisement( + second_node, + second_session, + second_endpoint, + &second_tls, + true, + true, + now_ms, + 15_000, + ), + now_ms, + ) + .await + .unwrap(); let enrolled = directory .create( advertisement( @@ -686,7 +881,7 @@ mod tests { .unwrap(); assert_eq!( enrolled.advertisement().log().unwrap().members(), - &[follower_node] + &[follower_node, second_node] ); let limits = Limits::default(); @@ -694,6 +889,18 @@ mod tests { let store = Arc::new( FollowerStore::open(store_root.clone(), limits, DiskBudget::new(1 << 30)).unwrap(), ); + let second_store_root = root.path().join("second-follower-store"); + let second_store = Arc::new( + FollowerStore::open(second_store_root.clone(), limits, DiskBudget::new(1 << 30)) + .unwrap(), + ); + let second_runtime = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 8).unwrap(), + 64 << 20, + second_session, + Host::default(), + ) + .unwrap(); let runtime = CellRuntime::new_with_replica_host( SqlWorkerPool::new(1, 8).unwrap(), 64 << 20, @@ -708,6 +915,33 @@ mod tests { follower_node, store.clone(), guard.clone(), + ) + .with_runtime_metrics(receiver_metrics.clone()); + let second_receiver = node_log_receiver::FollowerEndpoint::new( + directory.clone(), + second_runtime, + second_node, + second_store.clone(), + guard.clone(), + ) + .with_runtime_metrics(receiver_metrics.clone()); + let second_server = tokio::spawn(async move { + axum::serve( + second_tls.listener(second_listener), + node_log_receiver::router(second_receiver) + .into_make_service_with_connect_info::(), + ) + .await + }); + let duplicate = advertisement( + follower_node, + SessionId::from_bytes([8; 16]), + follower_endpoint, + &follower_tls, + true, + true, + now_ms, + 15_000, ); let follower_client = follower_tls.client_identity(); let server = tokio::spawn(async move { @@ -724,12 +958,15 @@ mod tests { leader_session, leader_node, guard.clone(), - ); + ) + .with_runtime_metrics(sender_metrics.clone()); let saved = frame(limits, leader_session); - for _ in 0..2 { - let receipt = transport - .append( - follower_node, + *discovery_store.counted_path.lock().unwrap() = + Some(layout.node_path(follower_session.as_bytes())); + let append_both = || { + futures_util::future::join_all([follower_node, second_node].map(|member| { + transport.append( + member, AppendRequest { leader_session, log_epoch: 2, @@ -737,10 +974,87 @@ mod tests { covered_through: 0, }, ) - .await - .unwrap(); - assert_eq!(receipt.durable_through, 1); + })) + }; + // Match the node-log shipper: one append to each enrolled member, + // with every receipt required before the batch can be acknowledged. + for receipt in append_both().await { + assert_eq!(receipt.unwrap().durable_through, 1); + } + let cold_reads = discovery_store.reads.load(Ordering::SeqCst); + for receipt in append_both().await { + assert_eq!(receipt.unwrap().durable_through, 1); + } + let warm_reads = discovery_store.reads.load(Ordering::SeqCst); + // Force only discovery TTL expiry; keep session/certificate/lease proof. + for peer in transport.peers.lock().unwrap().iter_mut() { + peer.verified_at_ms = unix_time_ms().unwrap() - PEER_CACHE_MS; } + for receipt in append_both().await { + assert_eq!(receipt.unwrap().durable_through, 1); + } + let refreshed_reads = discovery_store.reads.load(Ordering::SeqCst); + assert!(transport.discovery.lock().unwrap().upgrade().is_none()); + assert!(matches!( + transport.peer(NodeId::from_bytes([0; 16])).await, + Err(Error::Node("node identity is zero")) + )); + + // Cancel the initializer while a second follower lookup shares its + // cohort. The waiter must initialize afresh; nothing may stay detached. + transport.peers.lock().unwrap().clear(); + let entered = Arc::new(tokio::sync::Semaphore::new(0)); + *discovery_store.gate.lock().unwrap() = Some(DiscoveryGate { + entered: entered.clone(), + release: Arc::new(tokio::sync::Semaphore::new(0)), + }); + let first = { + let transport = transport.clone(); + tokio::spawn(async move { transport.peer(follower_node).await }) + }; + tokio::time::timeout(Duration::from_secs(2), entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let second = { + let transport = transport.clone(); + tokio::spawn(async move { transport.peer(second_node).await }) + }; + tokio::time::timeout(Duration::from_secs(2), async { + while transport.discovery.lock().unwrap().strong_count() < 2 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + first.abort(); + assert!(matches!(first.await, Err(error) if error.is_cancelled())); + let resumed = tokio::time::timeout(Duration::from_secs(2), second) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(resumed.member, second_node); + assert!(transport.discovery.lock().unwrap().upgrade().is_none()); + *discovery_store.counted_path.lock().unwrap() = None; + + // Keep resolve_node's fail-closed fleet validation even when the + // requested member is healthy and another node has duplicate sessions. + let duplicate = directory + .create(duplicate, unix_time_ms().unwrap()) + .await + .unwrap(); + transport.peers.lock().unwrap().clear(); + assert!(matches!( + transport.peer(second_node).await, + Err(Error::Node("multiple live sessions advertise one node")) + )); + assert!(transport.discovery.lock().unwrap().upgrade().is_none()); + directory + .withdraw_after_drain(&duplicate, unix_time_ms().unwrap()) + .await + .unwrap(); let local = PeerNodeLogTransport::new( directory.clone(), follower_client, @@ -800,11 +1114,36 @@ mod tests { .unwrap(); assert_eq!(page.frames, vec![saved.clone()]); assert_eq!(page.next_sequence, None); + for (metrics, phases) in [ + (&sender_metrics, ["peer_lookup", "round_trip"]), + (&receiver_metrics, ["enrollment", "durable_append"]), + ] { + let snapshot = metrics.snapshot(); + for phase in phases { + let observed = &snapshot["follower_append_phases"][phase]; + assert_eq!(observed["count"], 6, "{phase}: {observed}"); + assert_eq!(observed["failed"], 0); + assert_eq!(observed["cancelled"], 0); + assert_eq!(observed["in_flight"], 0); + } + } server.abort(); + second_server.abort(); let _ = server.await; + let _ = second_server.await; drop(transport); drop(local); drop(store); + drop(second_store); + let second_reopened = + FollowerStore::open(second_store_root, limits, DiskBudget::new(1 << 30)).unwrap(); + assert_eq!( + second_reopened + .read_tail(leader_session, 2, 1) + .await + .unwrap(), + vec![saved.clone()] + ); let reopened = FollowerStore::open(store_root, limits, DiskBudget::new(1 << 30)).unwrap(); assert_eq!( reopened @@ -818,6 +1157,16 @@ mod tests { reopened.read_tail(leader_session, 2, 1).await.unwrap(), vec![saved] ); + eprintln!( + "follower discovery reads: cold={cold_reads}, warm={warm_reads}, expired={refreshed_reads}" + ); + assert_eq!(cold_reads, 1, "concurrent cold appends repeated discovery"); + assert_eq!(warm_reads, cold_reads, "warm cache reloaded discovery"); + assert_eq!( + refreshed_reads, + cold_reads + 1, + "expired cache did not refresh" + ); } #[tokio::test] diff --git a/src/server/peer_receiver.rs b/src/server/peer_receiver.rs index 34f51de..a35ba76 100644 --- a/src/server/peer_receiver.rs +++ b/src/server/peer_receiver.rs @@ -47,12 +47,14 @@ pub(super) struct LocalResolver { registry: Arc, catalog_cache: Arc>>, handle_cache: Option>>>, + remote_owner_hints: Option>>>, provisioner: Option>, placement: Option>, bootstrap: Option, } const LOCAL_HANDLE_CACHE_TTL: Duration = Duration::from_millis(500); +const MAX_REMOTE_OWNER_HINTS: usize = 4_096; #[derive(Clone)] struct CachedHandle { @@ -79,6 +81,7 @@ impl LocalResolver { registry: peers.registry.clone(), catalog_cache: Arc::new(RwLock::new(HashMap::new())), handle_cache: handle_cache_enabled.then(|| Arc::new(RwLock::new(HashMap::new()))), + remote_owner_hints: handle_cache_enabled.then(|| Arc::new(RwLock::new(HashMap::new()))), provisioner: Some(provisioner), placement: None, bootstrap: None, @@ -106,7 +109,17 @@ impl LocalCellResolver for LocalResolver { if let Some(cache) = resolver.handle_cache.as_ref() { let cached = cache.read().await.get(&cell).cloned(); if let Some(cached) = cached { - if cached.expires_at > Instant::now() { + if cached.expires_at > Instant::now() + && resolver + .runtime + .active_handle(&target, CatalogRole::Sql) + .await? + .is_some_and(|active| { + active.owner_fence() == cached.handle.owner_fence() + && active.code() == cached.handle.code() + && active.schema() == cached.handle.schema() + }) + { return Ok(Some(cached.handle)); } cache.write().await.remove(&cell); @@ -157,9 +170,51 @@ impl LocalCellResolver for LocalResolver { { return Err(Error::CatalogCollision); } + if let Some(cache) = resolver.handle_cache.as_ref() + && let Some(local) = resolver + .runtime + .active_handle(&target, CatalogRole::Sql) + .await? + { + // The actor owns this capability and still fences drain, + // incarnation, code and schema at dispatch. Cache expiry + // need not read object storage for an active owner, including + // sparse owners while background hydration is still running. + cache.write().await.insert( + cell, + CachedHandle { + handle: local.clone(), + expires_at: Instant::now() + LOCAL_HANDLE_CACHE_TTL, + }, + ); + return Ok(Some(local)); + } let authority = CellAuthority::new(resolver.layout.clone()); for attempt in 0..2 { - let control = authority.load(target.cell_id()).await?; + let hinted_owner = if super::placement::is_placeable_target(&target) + && resolver.placement.is_some() + && let Some(hints) = &resolver.remote_owner_hints + { + hints.read().await.get(&cell).copied() + } else { + None + }; + let (control, owner_probe) = if let Some(hinted_owner) = hinted_owner + && let Some(placement) = &resolver.placement + { + // A session hint selects only which fresh canonical read to + // start early. It never supplies ownership or enrollment. + let observed_at_ms = unix_time_ms()?; + let (control, enrollment) = tokio::join!( + authority.load(cell), + placement + .directory + .load_if_live(hinted_owner, observed_at_ms), + ); + (control?, Some((hinted_owner, enrollment))) + } else { + (authority.load(cell).await?, None) + }; if let Some(control) = &control && let Some(local) = resolver .runtime @@ -185,7 +240,8 @@ impl LocalCellResolver for LocalResolver { .and_then(|control| control.value().owner.as_ref()) .map(|owner| owner.session); let expired = if let Some(placement) = &resolver.placement - && super::placement::is_placeable_target(&target) + && (super::placement::is_placeable_target(&target) + || target.namespace() == credentials::NAMESPACE) && let Some(control) = &control && control.value().root.is_some() && matches!( @@ -195,10 +251,47 @@ impl LocalCellResolver for LocalResolver { && let Some(owner) = owner && owner != placement.session { - !placement.directory.is_live(owner, unix_time_ms()?).await? + match owner_probe { + Some((hinted_owner, enrollment)) if hinted_owner == owner => { + let enrollment = enrollment?; + // Authority I/O may outlast the observed lease. A + // completed probe must still be live at this check. + let now_ms = unix_time_ms()?; + !enrollment.is_some_and(|enrollment| { + enrollment.advertisement().expires_at_ms() > now_ms + }) + } + // A changed owner requires its own fresh enrollment; + // errors for the obsolete hint are irrelevant. + _ => !placement.directory.is_live(owner, unix_time_ms()?).await?, + } } else { false }; + if let Some(hints) = &resolver.remote_owner_hints { + let mut hints = hints.write().await; + if super::placement::is_placeable_target(&target) + && !expired + && resolver.placement.as_ref().is_some_and(|placement| { + owner.is_some_and(|owner| owner != placement.session) + }) + && control.as_ref().is_some_and(|control| { + control.value().root.is_some() + && control.value().state == ControlState::Serving + }) + && let Some(owner) = owner + { + if !hints.contains_key(&cell) + && hints.len() >= MAX_REMOTE_OWNER_HINTS + && let Some(evicted) = hints.keys().next().copied() + { + hints.remove(&evicted); + } + hints.insert(cell, owner); + } else { + hints.remove(&cell); + } + } let needs_placement = expired || control.as_ref().is_some_and(|control| { control.value().root.is_some() @@ -228,6 +321,22 @@ impl LocalCellResolver for LocalResolver { } let control = control.ok_or(Error::CellNotActive)?; let resolved = async { + if expired + && target.namespace() == credentials::NAMESPACE + && let Some(placement) = &resolver.placement + { + // Authentication can outlive the node hosting its key + // shard. Restore only a cataloged published root; the + // takeover rechecks and fences the exact expired lease. + return provisioner + .takeover_expired_credential_cell(&target, &placement.directory) + .await + .map(Some) + .map_err(|source| Error::PeerTransport { + context: "BeyondDB credential owner recovery", + source: Box::new(source), + }); + } if needs_placement && super::placement::is_placeable_target(&target) && let Some(placement) = &resolver.placement @@ -342,6 +451,7 @@ struct Receiver { pub(super) fn peer_router( peers: &super::BeyonddbPeers, provisioner: Arc, + handle_cache_enabled: bool, ) -> Router { let runtime = peers.runtime.clone(); let directory = peers.placement.directory.clone(); @@ -352,7 +462,8 @@ pub(super) fn peer_router( layout: peers.layout.clone(), registry: peers.registry.clone(), catalog_cache: Arc::new(RwLock::new(HashMap::new())), - handle_cache: None, + handle_cache: handle_cache_enabled.then(|| Arc::new(RwLock::new(HashMap::new()))), + remote_owner_hints: None, // The sender selects ownership before forwarding. A receiver may // only dispatch to that active owner; a raced release must reject. provisioner: None, diff --git a/src/server/runtime_metrics.rs b/src/server/runtime_metrics.rs new file mode 100644 index 0000000..146dfbb --- /dev/null +++ b/src/server/runtime_metrics.rs @@ -0,0 +1,431 @@ +//! Optional bounded runtime observations for performance diagnosis. + +use std::{ + future::Future, + sync::atomic::{AtomicU64, Ordering}, + time::{Duration, Instant, SystemTime, UNIX_EPOCH}, +}; + +use cellule_runtime::fleet::telemetry::{ + ActivationPhase, CatalogReadKind, CellTelemetry, CommandResponseSource, + DurabilitySubmissionOutcome, PrimitiveOperationKind, PrimitiveOperationOutcome, + PublicationTiming, +}; +use cellule_runtime::identity::CellId; +use cellule_store::{StorageObservation, StorageObserver, StorageOperation, StorageOutcome}; +use serde_json::{Value, json}; + +#[derive(Default)] +struct Timing { + count: AtomicU64, + failed: AtomicU64, + total_us: AtomicU64, + max_us: AtomicU64, +} + +impl Timing { + fn record(&self, elapsed: Duration, succeeded: bool) { + let micros = elapsed.as_micros().min(u128::from(u64::MAX)) as u64; + self.total_us.fetch_add(micros, Ordering::Relaxed); + self.max_us.fetch_max(micros, Ordering::Relaxed); + if !succeeded { + self.failed.fetch_add(1, Ordering::Relaxed); + } + self.count.fetch_add(1, Ordering::Relaxed); + } + + fn snapshot(&self) -> Value { + json!({ + "count": self.count.load(Ordering::Relaxed), + "failed": self.failed.load(Ordering::Relaxed), + "total_us": self.total_us.load(Ordering::Relaxed), + "max_us": self.max_us.load(Ordering::Relaxed), + }) + } +} + +#[derive(Default)] +struct StoreOperationMetrics { + timing: Timing, + started: AtomicU64, + bytes_read: AtomicU64, + bytes_written: AtomicU64, + outcomes: [AtomicU64; StorageOutcome::ALL.len()], +} + +#[derive(Clone, Copy)] +pub(super) enum FollowerPhase { + PeerLookup = 0, + RoundTrip = 1, + Enrollment = 2, + DurableAppend = 3, +} + +#[derive(Default)] +struct FollowerPhaseMetrics { + timing: Timing, + started: AtomicU64, + cancelled: AtomicU64, +} + +impl FollowerPhaseMetrics { + fn snapshot(&self) -> Value { + let mut value = self.timing.snapshot(); + value["started"] = json!(self.started.load(Ordering::Relaxed)); + value["cancelled"] = json!(self.cancelled.load(Ordering::Relaxed)); + value["in_flight"] = json!( + self.started + .load(Ordering::Relaxed) + .saturating_sub(self.timing.count.load(Ordering::Relaxed)) + ); + value + } +} + +struct FollowerObservation<'a> { + metrics: &'a FollowerPhaseMetrics, + started: Instant, + outcome: Option, +} + +impl Drop for FollowerObservation<'_> { + fn drop(&mut self) { + if self.outcome.is_none() { + self.metrics.cancelled.fetch_add(1, Ordering::Relaxed); + } + self.metrics + .timing + .record(self.started.elapsed(), self.outcome == Some(true)); + } +} + +/// Observe only append phases; recovery operations are excluded. Dropped futures +/// count as cancelled failures. With metrics disabled, no clock or atomics run. +pub(super) async fn observe_follower( + metrics: Option<&RuntimeMetrics>, + phase: FollowerPhase, + future: impl Future>, +) -> cellule_runtime::Result { + let Some(metrics) = metrics else { + return future.await; + }; + let metrics = &metrics.follower_phases[phase as usize]; + metrics.started.fetch_add(1, Ordering::Relaxed); + let mut observation = FollowerObservation { + metrics, + started: Instant::now(), + outcome: None, + }; + let result = future.await; + observation.outcome = Some(result.is_ok()); + result +} + +impl StoreOperationMetrics { + fn snapshot(&self) -> Value { + let mut value = self.timing.snapshot(); + let started = self.started.load(Ordering::Relaxed); + let finished = self.timing.count.load(Ordering::Relaxed); + value["count"] = json!(finished); + value["started"] = json!(started); + value["in_flight"] = json!(started.saturating_sub(finished)); + value["bytes_read"] = json!(self.bytes_read.load(Ordering::Relaxed)); + value["bytes_written"] = json!(self.bytes_written.load(Ordering::Relaxed)); + value["outcomes"] = Value::Object( + StorageOutcome::ALL + .into_iter() + .map(|outcome| { + ( + outcome.label().to_owned(), + json!(self.outcomes[outcome.index()].load(Ordering::Relaxed)), + ) + }) + .collect(), + ); + value + } +} + +/// Fixed counters and timing sums; event callbacks perform no I/O or allocation. +/// +/// Snapshots observe concurrent atomics and are approximate, not a transactional +/// ledger. They include background runtime work. No Cell, tenant, or item IDs +/// appear as labels, and command response counts exclude queries and transport. +#[derive(Default)] +pub struct RuntimeMetrics { + first_snapshot_at_unix_ms: std::sync::OnceLock>, + response: [Timing; 3], + confirmation: Timing, + queue: Timing, + worker: Timing, + primitive: [Timing; 2], + publication: [Timing; 4], + submission: [AtomicU64; 4], + append: [AtomicU64; 2], + append_bytes: AtomicU64, + uploaded_objects: AtomicU64, + uploaded_bytes: AtomicU64, + catalog: [Timing; 2], + control: Timing, + activation: [Timing; 5], + store: [StoreOperationMetrics; StorageOperation::ALL.len()], + follower_phases: [FollowerPhaseMetrics; 4], +} + +impl RuntimeMetrics { + /// Returns a bounded JSON snapshot with durations in microseconds. + pub fn snapshot(&self) -> Value { + let sampled_at_unix_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .ok() + .map(|duration| duration.as_millis()); + json!({ + "version": 1, + "first_snapshot_at_unix_ms": self.first_snapshot_at_unix_ms.get_or_init(|| sampled_at_unix_ms), + "sampled_at_unix_ms": sampled_at_unix_ms, + "command_responses": { + "recorded": self.response[0].snapshot(), + "fleet": self.response[1].snapshot(), + "object": self.response[2].snapshot(), + }, + "confirmation": self.confirmation.snapshot(), + "command_queue": self.queue.snapshot(), + "command_worker": self.worker.snapshot(), + "primitive_execution": { + "command": self.primitive[0].snapshot(), + "query": self.primitive[1].snapshot(), + }, + "publication": { + "queue": self.publication[0].snapshot(), + "preparation": self.publication[1].snapshot(), + "authority": self.publication[2].snapshot(), + "total": self.publication[3].snapshot(), + "uploaded_objects": self.uploaded_objects.load(Ordering::Relaxed), + "uploaded_bytes": self.uploaded_bytes.load(Ordering::Relaxed), + }, + "durability_submissions": { + "fleet": self.submission[0].load(Ordering::Relaxed), + "unsupported": self.submission[1].load(Ordering::Relaxed), + "unavailable": self.submission[2].load(Ordering::Relaxed), + "rejected": self.submission[3].load(Ordering::Relaxed), + }, + "follower_appends": { + "acknowledged": self.append[0].load(Ordering::Relaxed), + "failed": self.append[1].load(Ordering::Relaxed), + "bytes": self.append_bytes.load(Ordering::Relaxed), + }, + "follower_append_phases": { + "peer_lookup": self.follower_phases[0].snapshot(), + "round_trip": self.follower_phases[1].snapshot(), + "enrollment": self.follower_phases[2].snapshot(), + "durable_append": self.follower_phases[3].snapshot(), + }, + "catalog_reads": {"head": self.catalog[0].snapshot(), "page": self.catalog[1].snapshot()}, + "control_reads": self.control.snapshot(), + "object_store": StorageOperation::ALL.into_iter().map(|operation| { + (operation.label().to_owned(), self.store[operation.index()].snapshot()) + }).collect::>(), + "activation": { + "ownership": self.activation[0].snapshot(), + "resume": self.activation[1].snapshot(), + "root_open": self.activation[2].snapshot(), + "restore": self.activation[3].snapshot(), + "activate": self.activation[4].snapshot(), + }, + }) + } +} + +impl CellTelemetry for RuntimeMetrics { + fn command_response( + &self, + source: CommandResponseSource, + elapsed: Duration, + confirmation: Duration, + ) { + let index = match source { + CommandResponseSource::Recorded => 0, + CommandResponseSource::Fleet => 1, + CommandResponseSource::Object => 2, + }; + self.response[index].record(elapsed, true); + self.confirmation.record(confirmation, true); + } + + fn command_execution(&self, queue: Duration, worker: Duration, succeeded: bool) { + self.queue.record(queue, succeeded); + self.worker.record(worker, succeeded); + } + + fn primitive_operation( + &self, + _module: &'static str, + kind: PrimitiveOperationKind, + outcome: PrimitiveOperationOutcome, + elapsed: Duration, + ) { + let index = match kind { + PrimitiveOperationKind::Command => 0, + PrimitiveOperationKind::Query => 1, + }; + self.primitive[index].record(elapsed, outcome != PrimitiveOperationOutcome::Failed); + } + + fn publication_completed(&self, _cell: CellId, timing: PublicationTiming) { + for (metric, elapsed) in self.publication.iter().zip([ + timing.queue_wait, + timing.preparation, + timing.authority, + timing.total, + ]) { + metric.record(elapsed, timing.succeeded); + } + } + + fn durability_submission(&self, outcome: DurabilitySubmissionOutcome) { + let index = match outcome { + DurabilitySubmissionOutcome::Fleet => 0, + DurabilitySubmissionOutcome::Unsupported => 1, + DurabilitySubmissionOutcome::Unavailable => 2, + DurabilitySubmissionOutcome::Rejected => 3, + }; + self.submission[index].fetch_add(1, Ordering::Relaxed); + } + + fn publication_cost(&self, objects: u64, bytes: u64) { + self.uploaded_objects.fetch_add(objects, Ordering::Relaxed); + self.uploaded_bytes.fetch_add(bytes, Ordering::Relaxed); + } + + fn node_log_append(&self, acknowledged: bool, bytes: u64) { + self.append[usize::from(!acknowledged)].fetch_add(1, Ordering::Relaxed); + self.append_bytes.fetch_add(bytes, Ordering::Relaxed); + } + + fn catalog_read(&self, kind: CatalogReadKind, elapsed: Duration, succeeded: bool) { + let index = match kind { + CatalogReadKind::Head => 0, + CatalogReadKind::Page => 1, + }; + self.catalog[index].record(elapsed, succeeded); + } + + fn control_read(&self, elapsed: Duration, succeeded: bool) { + self.control.record(elapsed, succeeded); + } + + fn activation_phase(&self, phase: ActivationPhase, elapsed: Duration) { + let index = match phase { + ActivationPhase::Ownership => 0, + ActivationPhase::Resume => 1, + ActivationPhase::RootOpen => 2, + ActivationPhase::Restore => 3, + ActivationPhase::Activate => 4, + }; + self.activation[index].record(elapsed, true); + } +} + +impl StorageObserver for RuntimeMetrics { + fn started(&self, operation: StorageOperation) { + self.store[operation.index()] + .started + .fetch_add(1, Ordering::Relaxed); + } + + fn finished(&self, observation: StorageObservation) { + let metric = &self.store[observation.operation.index()]; + metric + .bytes_read + .fetch_add(observation.bytes_read, Ordering::Relaxed); + metric + .bytes_written + .fetch_add(observation.bytes_written, Ordering::Relaxed); + metric.outcomes[observation.outcome.index()].fetch_add(1, Ordering::Relaxed); + metric.timing.record( + observation.duration, + observation.outcome == StorageOutcome::Success, + ); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use cellule_store::Store; + use object_store::{ObjectStoreExt, memory::InMemory, path::Path}; + use std::sync::Arc; + use tokio_util::bytes::Bytes; + + #[tokio::test] + async fn follower_metrics_finish_when_an_in_flight_append_is_cancelled() { + let metrics = RuntimeMetrics::default(); + let observed = observe_follower( + Some(&metrics), + FollowerPhase::DurableAppend, + std::future::pending::>(), + ); + let mut observed = Box::pin(observed); + assert!(futures_util::poll!(&mut observed).is_pending()); + let snapshot = metrics.snapshot(); + assert_eq!( + snapshot["follower_append_phases"]["durable_append"]["in_flight"], + 1 + ); + drop(observed); + let snapshot = metrics.snapshot(); + let phase = &snapshot["follower_append_phases"]["durable_append"]; + assert_eq!(phase["count"], 1); + assert_eq!(phase["failed"], 1); + assert_eq!(phase["cancelled"], 1); + assert_eq!(phase["in_flight"], 0); + } + + #[tokio::test] + async fn store_metrics_observe_consumption_cancellation_and_missing_objects_without_labels() { + let metrics = Arc::new(RuntimeMetrics::default()); + let store = Store::new(Arc::new(InMemory::new())).with_storage_observer(metrics.clone()); + let path = Path::from("private-account/never-a-metric-label"); + store + .inner() + .put(&path, Bytes::from_static(b"abc").into()) + .await + .unwrap(); + let body = store + .inner() + .get(&path) + .await + .unwrap() + .bytes() + .await + .unwrap(); + assert_eq!(body.as_ref(), b"abc"); + // A returned body is still an active operation until consumed or dropped. + let unconsumed = store.inner().get(&path).await.unwrap(); + assert_eq!(metrics.snapshot()["object_store"]["get"]["in_flight"], 1); + drop(unconsumed); + assert!( + store + .inner() + .head(&Path::from("missing-private-object")) + .await + .is_err() + ); + let snapshot = metrics.snapshot(); + let get = &snapshot["object_store"]["get"]; + assert_eq!(get["count"], 2); + assert_eq!(get["failed"], 1); + assert_eq!(get["bytes_read"], 3); + assert_eq!(get["in_flight"], 0); + assert_eq!(get["outcomes"]["success"], 1); + assert_eq!(get["outcomes"]["cancelled"], 1); + let put = &snapshot["object_store"]["put"]; + assert_eq!(put["count"], 1); + assert_eq!(put["bytes_written"], 3); + assert_eq!(snapshot["object_store"]["head"]["outcomes"]["not_found"], 1); + let serialized = snapshot.to_string(); + assert!(!serialized.contains("private-account")); + assert!(!serialized.contains("never-a-metric-label")); + assert!(!serialized.contains("missing-private-object")); + } +} diff --git a/src/stream_retention.rs b/src/stream_retention.rs index 0c66c3f..a6c229d 100644 --- a/src/stream_retention.rs +++ b/src/stream_retention.rs @@ -157,7 +157,8 @@ mod tests { use cellule_app::CellApplication; use cellule_ltx::rusqlite::{Connection, params}; use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, WireValue}; - use cellule_runtime::identity::{CellTarget, Digest, TenantId}; + use cellule_runtime::control::OwnerFence; + use cellule_runtime::identity::{CellTarget, Digest, IncarnationId, TenantId}; use cellule_runtime::registry::{BuildDescriptor, CommandInvocation, QueryInvocation}; use super::*; @@ -249,6 +250,10 @@ mod tests { codec_version: 1, schema: 1, target: target.clone(), + owner_fence: OwnerFence { + incarnation: IncarnationId::from_bytes([1; 16]), + epoch: 1, + }, sequence, now_ms: RETENTION_MS + 1_000, input: &input, diff --git a/src/table.rs b/src/table.rs index fa08c02..78cf018 100644 --- a/src/table.rs +++ b/src/table.rs @@ -38,10 +38,28 @@ impl TryFrom<&str> for TableClass { pub enum TablePlacement { /// Items are stored directly in the account Cell. Account, + /// One dedicated base data Cell, with automatic and manual base splits disabled. + Single, /// Base and index directories start with this many ranges each. Routed { initial_partitions: u16 }, } +impl TablePlacement { + /// Initial base and global-index range count; account-local tables have no ranges. + pub const fn initial_partitions(self) -> Option { + match self { + Self::Account => None, + Self::Single => Some(1), + Self::Routed { initial_partitions } => Some(initial_partitions), + } + } + + /// Whether data lives outside the account metadata Cell. + pub const fn is_routed(self) -> bool { + self.initial_partitions().is_some() + } +} + /// An ExtendDB table's key contract stored in the account Cell. #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] pub struct TableSpec { @@ -237,7 +255,7 @@ impl Command for CreateTable { ], ))?; } - if matches!(record.placement, TablePlacement::Routed { .. }) { + if record.placement.is_routed() { // Persist lifecycle ownership before provisioning independent roots. // Deletion must fence even an installer that never publishes its copy. for id in std::iter::once(&record.id).chain( @@ -371,7 +389,7 @@ impl Command for UpdateTable { }; // Initial owners persist this record verbatim. Keep it immutable until // route publication so a retry cannot conflict with installed owners. - if matches!(table.placement, TablePlacement::Routed { .. }) + if table.placement.is_routed() && context.sql(&statement( "SELECT 1 FROM ddb_directory_roots WHERE table_id = ?1 AND initial_fingerprint IS NOT NULL", vec![SqlValue::Text(table.id.clone())], diff --git a/src/transaction_coordinator.rs b/src/transaction_coordinator.rs index d80552d..d4fdad7 100644 --- a/src/transaction_coordinator.rs +++ b/src/transaction_coordinator.rs @@ -37,7 +37,7 @@ static NAMESPACES: [NamespaceDescriptor; 1] = [NamespaceDescriptor { effect_targets: &[], dead_letter: None, }]; -static COMMANDS: [OperationDescriptor; 10] = [ +static COMMANDS: [OperationDescriptor; 11] = [ operation(1), crate::participant::phase_operation(2), operation(3), @@ -60,8 +60,13 @@ static COMMANDS: [OperationDescriptor; 10] = [ output_limit: 64 * 1024, ..operation(10) }, + OperationDescriptor { + input_limit: 64 * 1024, + output_limit: 64 * 1024, + ..operation(11) + }, ]; -static QUERIES: [OperationDescriptor; 6] = [ +static QUERIES: [OperationDescriptor; 7] = [ OperationDescriptor { codec_version: 2, ..operation(1) @@ -78,6 +83,11 @@ static QUERIES: [OperationDescriptor; 6] = [ }, operation(5), operation(6), + OperationDescriptor { + input_limit: 4096, + output_limit: resume::RESUME_OUTPUT_BYTES, + ..operation(7) + }, ]; const fn operation(id: u32) -> OperationDescriptor { @@ -110,6 +120,7 @@ impl cellule_runtime::registry::CellModule for CoordinatorModule { source.update(include_bytes!("transaction_coordinator/phase.rs")); source.update(include_bytes!("transaction_coordinator/read_release.rs")); source.update(include_bytes!("transaction_coordinator/token.rs")); + source.update(include_bytes!("transaction_coordinator/resume.rs")); source.update(include_bytes!("transaction_token.rs")); source.update(include_bytes!("items.rs")); source.update(include_bytes!("expression_wire.rs")); @@ -137,6 +148,7 @@ impl cellule_runtime::registry::CellModule for CoordinatorModule { registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; + registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; registry.bind_command::()?; @@ -147,6 +159,7 @@ impl cellule_runtime::registry::CellModule for CoordinatorModule { registry.bind_query::()?; registry.bind_query::()?; registry.bind_query::()?; + registry.bind_query::()?; registry.bind_query::() } } @@ -473,8 +486,10 @@ fn read_decision( mod phase; mod read_release; mod registry; +mod resume; mod token; pub use phase::*; pub use read_release::*; pub use registry::*; +pub use resume::*; pub use token::*; diff --git a/src/transaction_coordinator/phase.rs b/src/transaction_coordinator/phase.rs index 8e4d55c..183f3aa 100644 --- a/src/transaction_coordinator/phase.rs +++ b/src/transaction_coordinator/phase.rs @@ -304,6 +304,137 @@ impl Command for DecideCrossCellTransaction { } } +/// One durable participant receipt, scoped by the enclosing transaction. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct CoordinatorPrepareReceipt { + pub position: u8, + pub participant_cell: [u8; 32], + pub sequence: u64, +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct CommitPreparedTransactionInput { + pub transaction: ReadCrossCellTransactionInput, + pub prepares: Vec, +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub enum CommitPreparedTransactionOutcome { + EvidenceRejected(Vec), + Decision(DecideCrossCellTransactionOutcome), +} + +/// Publish prepare evidence and the terminal decision in one Cell command. +/// +/// The existing decision handler still checks every durable participant row. +/// A partial receipt set cannot commit an unprepared transaction. Older drivers +/// and recovery can continue using the separate phase commands. +pub struct CommitPreparedTransaction; + +impl Command for CommitPreparedTransaction { + const MODULE: &'static str = MODULE; + const ID: u32 = 11; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + let transaction = input.transaction; + phase_identity(context, &transaction.account_id, &transaction.routing_key)?; + if input.prepares.len() > 100 { + return Err(Error::Command("prepare evidence count exceeds 100")); + } + let mut outcomes = Vec::with_capacity(input.prepares.len()); + for receipt in input.prepares { + let result = record_participant_prepare( + context, + CoordinatorPhaseInput { + account_id: transaction.account_id.clone(), + transaction_id: transaction.transaction_id, + routing_key: transaction.routing_key.clone(), + position: receipt.position, + participant_cell: receipt.participant_cell, + sequence: receipt.sequence, + }, + )?; + outcomes.push(match result { + CommandResult::Success(Json(outcome)) | CommandResult::Rejected(Json(outcome)) => { + outcome + } + }); + } + if outcomes.iter().any(|outcome| { + !matches!( + outcome, + CoordinatorPhaseOutcome::Recorded | CoordinatorPhaseOutcome::Replay + ) + }) { + return Ok(CommandResult::Rejected(Json( + CommitPreparedTransactionOutcome::EvidenceRejected(outcomes), + ))); + } + let decision = DecideCrossCellTransaction::execute( + context, + Json(DecideCrossCellTransactionInput { + account_id: transaction.account_id, + transaction_id: transaction.transaction_id, + routing_key: transaction.routing_key, + decision: CoordinatorDecision::Commit, + }), + )?; + Ok(match decision { + CommandResult::Success(Json(outcome)) => { + CommandResult::Success(Json(CommitPreparedTransactionOutcome::Decision(outcome))) + } + CommandResult::Rejected(Json(outcome)) => { + CommandResult::Rejected(Json(CommitPreparedTransactionOutcome::Decision(outcome))) + } + }) + } +} + +#[cfg(test)] +mod prepared_commit_tests { + use super::*; + + #[test] + fn maximum_prepare_receipts_fit_the_registered_wire_limit() { + let input = Json(CommitPreparedTransactionInput { + transaction: ReadCrossCellTransactionInput { + account_id: "123456789012".into(), + transaction_id: [255; 16], + routing_key: vec![255; 128], + }, + prepares: (0..100) + .map(|position| CoordinatorPrepareReceipt { + position, + participant_cell: [255; 32], + sequence: i64::MAX as u64, + }) + .collect(), + }); + let descriptor = super::super::COMMANDS + .iter() + .find(|operation| operation.id == CommitPreparedTransaction::ID) + .unwrap(); + let limit = descriptor.input_limit; + let mut encoder = BoundedEncoder::new(limit).unwrap(); + input.encode(&mut encoder).unwrap(); + let encoded = encoder.finish(); + let mut decoder = BoundedDecoder::new(&encoded, limit).unwrap(); + assert_eq!( + Json::::decode(&mut decoder) + .unwrap() + .0, + input.0 + ); + decoder.finish().unwrap(); + } +} + /// Record evidence that one participant resolution was published. pub struct RecordParticipantResolution; diff --git a/src/transaction_coordinator/registry.rs b/src/transaction_coordinator/registry.rs index a568a7b..8692f5a 100644 --- a/src/transaction_coordinator/registry.rs +++ b/src/transaction_coordinator/registry.rs @@ -7,6 +7,8 @@ use super::SHARDS; use crate::table::statement; use crate::{Error, Json, MODULE, Result, SqlValue, account_target}; +pub(crate) const MAX_REGISTRATION_SHARDS: usize = 64; + #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] pub struct RegisterCoordinatorShardInput { pub account_id: String, @@ -43,6 +45,47 @@ impl Command for RegisterCoordinatorShard { } } +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct RegisterCoordinatorShardsInput { + pub account_id: String, + pub shards: Vec, +} + +/// Publish several coordinator registrations in one durable account command. +pub struct RegisterCoordinatorShards; + +impl Command for RegisterCoordinatorShards { + const MODULE: &'static str = MODULE; + const ID: u32 = 56; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json<()>; + + fn execute( + context: &mut CommandContext<'_, '_>, + Json(input): Self::Input, + ) -> Result> { + if account_target(&input.account_id)? != *context.target() { + return Err(Error::Identity( + "coordinator shards reached the wrong account", + )); + } + if input.shards.is_empty() + || input.shards.len() > MAX_REGISTRATION_SHARDS + || input.shards.iter().any(|shard| *shard >= SHARDS) + { + return Err(Error::Command("invalid coordinator registration batch")); + } + for shard in input.shards { + context.sql(&statement( + "INSERT OR IGNORE INTO ddb_coordinator_shards (shard) VALUES (?1)", + vec![SqlValue::Integer(i64::from(shard))], + ))?; + } + Ok(CommandResult::Success(Json(()))) + } +} + #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] pub struct ListCoordinatorShardsInput { pub account_id: String, diff --git a/src/transaction_coordinator/resume.rs b/src/transaction_coordinator/resume.rs new file mode 100644 index 0000000..3670f25 --- /dev/null +++ b/src/transaction_coordinator/resume.rs @@ -0,0 +1,120 @@ +//! Bounded preparation discovery from one coordinator observation. + +use cellule_runtime::registry::{Query, QueryContext}; +use serde::{Deserialize, Serialize}; + +use super::{ + CoordinatorDecision, CoordinatorParticipant, CrossCellTransactionStatus, MODULE, + ReadCoordinatorParticipant, ReadCoordinatorParticipantInput, ReadCrossCellTransaction, + ReadCrossCellTransactionInput, +}; +use crate::{Error, Json, Result, SqlValue, table::statement}; + +pub(super) const RESUME_OUTPUT_BYTES: u32 = 64 * 1024; +const INLINE_PAYLOAD_BYTES: i64 = 32 * 1024; + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct CoordinatorResumeParticipant { + pub position: u8, + pub participant: CoordinatorParticipant, +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +pub struct CoordinatorResumeSnapshot { + pub status: Option, + /// `None` retains chunked discovery; `Some([])` means no prepares remain. + pub participants: Option>, +} + +/// Read status and small, unprepared immutable payloads through one Cell query. +/// +/// Large inputs retain the existing chunk protocol. Terminal observations +/// contain no payloads: only their durable decision permits resolution. +pub struct ReadCoordinatorResume; + +impl Query for ReadCoordinatorResume { + const MODULE: &'static str = MODULE; + const ID: u32 = 7; + const CODEC_VERSION: u32 = 1; + type Input = Json; + type Output = Json; + + fn execute(context: &mut QueryContext<'_>, Json(input): Self::Input) -> Result { + let status = ReadCrossCellTransaction::execute(context, Json(input.clone()))?.0; + let mut snapshot = CoordinatorResumeSnapshot { + status, + participants: None, + }; + if snapshot + .status + .as_ref() + .is_none_or(|status| status.decision != CoordinatorDecision::Begin) + { + return Ok(Json(snapshot)); + } + // Inspect lengths before reading any blob. One handler runs under the + // Cell's query observation; commands cannot compact between these reads. + // Prepared participants already have durable evidence and need no payload. + let rows = context.sql(&statement( + "SELECT p.position, p.operation_chunks, COUNT(c.chunk), SUM(length(c.payload)) \ + FROM ddb_coordinator_participants p \ + LEFT JOIN ddb_transaction_payloads c ON c.transaction_id = p.transaction_id \ + AND c.position = p.position \ + WHERE p.transaction_id = ?1 AND p.prepared_sequence IS NULL \ + AND p.resolved_sequence IS NULL GROUP BY p.position ORDER BY p.position", + vec![SqlValue::Blob(input.transaction_id.to_vec())], + ))?; + let mut positions = Vec::with_capacity(rows[0].rows.len()); + let mut bytes = 0_i64; + for row in &rows[0].rows { + let [ + SqlValue::Integer(position), + SqlValue::Integer(chunks), + SqlValue::Integer(stored), + SqlValue::Integer(size), + ] = row.as_slice() + else { + return Ok(Json(snapshot)); + }; + if *chunks != 1 || *stored != 1 || *size <= 0 || *size > INLINE_PAYLOAD_BYTES - bytes { + return Ok(Json(snapshot)); + } + bytes += size; + positions.push( + u8::try_from(*position) + .map_err(|_| Error::Command("invalid coordinator participant position"))?, + ); + } + let mut participants = Vec::with_capacity(positions.len()); + for position in positions { + let Some(chunk) = ReadCoordinatorParticipant::execute( + context, + Json(ReadCoordinatorParticipantInput { + account_id: input.account_id.clone(), + transaction_id: input.transaction_id, + routing_key: input.routing_key.clone(), + position, + chunk: 0, + }), + )? + else { + return Ok(Json(snapshot)); + }; + participants.push(CoordinatorResumeParticipant { + position, + participant: CoordinatorParticipant { + target: chunk.target, + operations: serde_json::from_slice(&chunk.payload)?, + }, + }); + } + snapshot.participants = Some(participants); + // The envelope and canonical JSON can exceed the raw payload sum. + // Leave room for WireValue's four-byte length prefix and fall back + // before transport admission rather than failing a valid large request. + if serde_json::to_vec(&snapshot)?.len() > RESUME_OUTPUT_BYTES as usize - 4 { + snapshot.participants = None; + } + Ok(Json(snapshot)) + } +} diff --git a/tests/elastic_cells.rs b/tests/elastic_cells.rs index 44f5d10..180741e 100644 --- a/tests/elastic_cells.rs +++ b/tests/elastic_cells.rs @@ -50,12 +50,15 @@ mod elastic_cells { mod read_release; mod read_resolution; mod recovery_admission; + mod table_key_cache; mod table_residency; mod transaction_capacity; + mod transaction_commit; mod transaction_driver; mod transaction_reads; mod transaction_recovery; mod transaction_resolution; + mod transaction_resume; mod transaction_transport; pub(crate) mod transaction_visibility; mod transaction_write_skew; diff --git a/tests/elastic_cells/public_transactions.rs b/tests/elastic_cells/public_transactions.rs index 2792788..afdc727 100644 --- a/tests/elastic_cells/public_transactions.rs +++ b/tests/elastic_cells/public_transactions.rs @@ -19,36 +19,98 @@ pub(super) async fn assert_lost_replies_and_canceled_token_reuse( registry.release_digest(), SigningKey::from_bytes(&[221; 32]), ); - let lost = Arc::new(std::sync::atomic::AtomicU8::new(0)); - let transport = DropPhaseReplies { - verifier: Arc::new(PeerVerifier::new( - session, - registry.release_digest(), - signer.verifying_key(), - )), - dispatcher: Arc::new(PeerDispatcher::new( + let signer = Arc::new(signer); + let verifier = Arc::new(PeerVerifier::new( + session, + registry.release_digest(), + signer.verifying_key(), + )); + let dispatcher = Arc::new(PeerDispatcher::new( + registry.clone(), + Arc::new(LocalRuntimePeerResolver { + runtime: owner.runtime(), + layout: layout.clone(), + }), + Arc::new(TestPeerAuthorizer), + )); + let phase_client = |enabled, lost| { + CellClient::runtime_with_peer( registry.clone(), - Arc::new(LocalRuntimePeerResolver { - runtime: owner.runtime(), - layout: layout.clone(), + runtime.clone(), + layout.clone(), + signer.clone(), + PeerPrincipal { + issuer: "admission-test".into(), + subject: "admission".into(), + actions: vec!["beyonddb.cell.invoke".into()], + }, + Arc::new(DropPhaseReplies { + verifier: verifier.clone(), + dispatcher: dispatcher.clone(), + lost, + enabled, }), - Arc::new(TestPeerAuthorizer), - )), - lost: lost.clone(), - enabled: 127, + ) }; - let client = CellClient::runtime_with_peer( - registry, - runtime.clone(), - layout, - Arc::new(signer), - PeerPrincipal { - issuer: "admission-test".into(), - subject: "admission".into(), - actions: vec!["beyonddb.cell.invoke".into()], - }, - Arc::new(transport), - ); + // Drop only completion replies so BEGIN and COMMIT are acknowledged and + // the fresh-write shortcut is exercised. A missing participant reply can + // be recovered by observing its durable state; a missing coordinator + // receipt must leave the first call uncertain until token replay. + for phase in [8, 128] { + let dropped = Arc::new(std::sync::atomic::AtomicU8::new(0)); + let fresh = CellStorage::new(phase_client(phase, dropped.clone()), "us-east-1") + .with_transaction_coordinators(provisioner.clone()); + let item = Item::from([ + ( + "id".into(), + AttributeValue::S(format!("fresh-completion-{phase}")), + ), + ("value".into(), AttributeValue::N("7".into())), + ]); + let maps = ExpressionMaps::default(); + let ops = infos + .iter() + .map(|info| TransactWriteOp::Put { + key_info: info, + item: &item, + condition: None, + maps: &maps, + return_values_on_ccf: Default::default(), + stream: None, + }) + .collect::>(); + let token_text = format!("fresh-completion-{phase}"); + let token = || IdempotencyKey { + account_id: "123456789012", + token: &token_text, + fingerprint: "two-participant-write", + }; + let result = fresh.transact_write_items(&ops, Some(token())).await; + if phase == 8 { + result.unwrap(); + } else { + assert!( + matches!(result, Err(StorageError::Transient(_))), + "lost coordinator receipt cannot prove completion: {result:?}" + ); + } + assert_eq!(dropped.load(Ordering::SeqCst), phase); + assert!(matches!( + fresh.transact_write_items(&ops, Some(token())).await, + Err(StorageError::IdempotentReplay) + )); + for info in &infos { + assert_eq!( + fresh + .get_item(info, &Item::from([("id".into(), item["id"].clone())])) + .await + .unwrap(), + Some(item.clone()) + ); + } + } + let lost = Arc::new(std::sync::atomic::AtomicU8::new(0)); + let client = phase_client(127, lost.clone()); // The adapter must keep transaction decisions and intent barriers on the // owner even when its caller supplies a snapshot-read capability. let storage = CellStorage::new( @@ -200,6 +262,86 @@ pub(super) async fn assert_lost_replies_and_canceled_token_reuse( &infos, ) .await; + // Exercise bounded prepares for both legacy account and routed data + // participants. A small conditional request can return a much larger old + // image. Losing its bounded rejection must stay uncertain, not fall back. + for (index, info) in infos.iter().enumerate() { + let old_key = Item::from([( + "id".into(), + AttributeValue::S(format!("wide-failure-{index}")), + )]); + let mut old = old_key.clone(); + old.insert("payload".into(), AttributeValue::S("\0".repeat(384 * 1024))); + storage + // Seed through the wide command. The legacy account no-return + // PutItem envelope is 1 MiB; JSON escaping expands this valid item. + .put_item(info, old.clone(), true, None, &maps, None) + .await + .unwrap(); + let new = Item::from([( + "id".into(), + AttributeValue::S(format!("wide-rollback-{index}")), + )]); + let operations = [ + TransactWriteOp::Put { + key_info: info, + item: &new, + condition: None, + maps: &maps, + return_values_on_ccf: Default::default(), + stream: None, + }, + TransactWriteOp::Put { + key_info: info, + item: &old_key, + condition: Some(¬_exists), + maps: &maps, + return_values_on_ccf: ReturnValuesOnConditionCheckFailure::AllOld, + stream: None, + }, + ]; + let dropped = Arc::new(std::sync::atomic::AtomicU8::new(0)); + let uncertain = CellStorage::new(phase_client(1, dropped.clone()), "us-east-1") + .with_transaction_coordinators(provisioner.clone()); + let token_value = format!("wide-condition-{index}"); + let token = || IdempotencyKey { + account_id: "123456789012", + token: &token_value, + fingerprint: "wide-failure-image", + }; + assert!(matches!( + uncertain + .transact_write_items(&operations, Some(token())) + .await, + Err(StorageError::Transient(_)) + )); + assert_eq!(dropped.load(Ordering::SeqCst), 1); + assert_eq!(storage.get_item(info, &new).await.unwrap(), None); + assert_eq!( + storage.get_item(info, &old_key).await.unwrap(), + Some(old.clone()) + ); + assert!(matches!( + uncertain.transact_write_items(&operations, Some(token())).await, + Err(StorageError::TransactionCanceled(reasons)) + if reasons[1].code == "ConditionalCheckFailed" && reasons[1].item == Some(old.clone()) + )); + assert_eq!(storage.get_item(info, &new).await.unwrap(), None); + // No prepared locks remain after the rejected attempt and ABORT. + storage + .delete_item(info, &old_key, false, None, &maps, None) + .await + .unwrap(); + uncertain + .transact_write_items(&operations, Some(token())) + .await + .unwrap(); + assert_eq!(storage.get_item(info, &new).await.unwrap(), Some(new)); + assert_eq!( + storage.get_item(info, &old_key).await.unwrap(), + Some(old_key) + ); + } runtime.shutdown().await.unwrap(); } diff --git a/tests/elastic_cells/recovery_admission.rs b/tests/elastic_cells/recovery_admission.rs index b10804f..6c753a0 100644 --- a/tests/elastic_cells/recovery_admission.rs +++ b/tests/elastic_cells/recovery_admission.rs @@ -179,6 +179,13 @@ async fn recovery_case(startup: bool, decision: CoordinatorDecision) { let blocked = layout.control_path(participants[0].0.cell_id().as_bytes()); objects.reset(); objects.block_body_reads_for(&blocked); + // Recovery must encounter the admission fault with cold routing state. + // A client warmed by prepare can reuse a verified local route and reach + // the live actor without reading the control object blocked above. + // Keep the fixed-handle observer so assertions can still inspect that actor. + let client = CellClient::local_runtime(registry, host.runtime(), layout.clone()); + let storage = + CellStorage::new(client.clone(), "us-east-1").with_initial_partitions(provisioner.clone()); if startup { assert!( provisioner diff --git a/tests/elastic_cells/table_key_cache.rs b/tests/elastic_cells/table_key_cache.rs new file mode 100644 index 0000000..c13cb27 --- /dev/null +++ b/tests/elastic_cells/table_key_cache.rs @@ -0,0 +1,173 @@ +use crate::*; + +use cellule_runtime::fleet::telemetry::{ + CellTelemetry, PrimitiveOperationKind, PrimitiveOperationOutcome, +}; + +#[derive(Default)] +struct AccountQueries(std::sync::atomic::AtomicU64); + +impl CellTelemetry for AccountQueries { + fn primitive_operation( + &self, + module: &'static str, + kind: PrimitiveOperationKind, + _: PrimitiveOperationOutcome, + _: std::time::Duration, + ) { + if module == "beyonddb-account" && kind == PrimitiveOperationKind::Query { + self.0.fetch_add(1, Ordering::Relaxed); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn backend_key_info_cache_reduces_queries_and_fences_recreated_generations() { + let application = Arc::new( + Beyonddb::compile(BuildDescriptor { + source_revision: "backend-key-info-cache".into(), + cargo_lock_digest: Digest::from_bytes([1; 32]), + }) + .unwrap(), + ); + let account_id = "123456789012"; + let account = account_target(account_id).unwrap(); + let session = SessionId::from_bytes([97; 16]); + let files = tempfile::tempdir().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + object_store::path::Path::from("key-info-cache"), + *account.application().as_bytes(), + ); + let host = CellNodeBuilder::new(application.clone()) + .with_runtime(SqlWorkerPool::new(1, 8).unwrap(), 16 << 20) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(1 << 30))) + .with_session(session) + .build_unleased_for_maintenance() + .unwrap(); + let queries = Arc::new(AccountQueries::default()); + host.install_telemetry(queries.clone()).unwrap(); + let registry = application.registry(); + let bootstrap = Bootstrap { + runtime: host.runtime(), + registry: ®istry, + layout: &layout, + session, + }; + let handle = bootstrap + .cell( + &account, + "beyonddb-account", + 98, + &files.path().join("account.sqlite"), + initialize_account, + ) + .await; + let client = + CellClient::local_with_telemetry(registry, handle, host.runtime().telemetry_handle()); + let cached = CellStorage::new(client.clone(), "us-east-1").with_route_cache(true); + let fresh = CellStorage::new(client, "us-east-1"); + let create = || CreateTableInput { + table_name: "KeyInfo".into(), + billing_mode: Some(BillingMode::PayPerRequest), + key_schema: vec![KeySchemaElement { + attribute_name: "pk".into(), + key_type: KeyType::Hash, + }], + attribute_definitions: vec![AttributeDefinition { + attribute_name: "pk".into(), + attribute_type: ScalarAttributeType::S, + }], + ..Default::default() + }; + cached.create_table(account_id, create()).await.unwrap(); + let first = cached.table_key_info(account_id, "KeyInfo").await.unwrap(); + let before = queries.0.load(Ordering::Relaxed); + assert_eq!( + cached + .table_key_info(account_id, "KeyInfo") + .await + .unwrap() + .table_id, + first.table_id + ); + assert_eq!( + queries.0.load(Ordering::Relaxed), + before, + "warm backend metadata lookup must not execute another Cell query" + ); + fresh.table_key_info(account_id, "KeyInfo").await.unwrap(); + fresh.table_key_info(account_id, "KeyInfo").await.unwrap(); + assert_eq!( + queries.0.load(Ordering::Relaxed), + before + 2, + "the default mode must retain fresh metadata reads" + ); + + cached + .update_table( + account_id, + serde_json::from_value(serde_json::json!({ + "TableName": "KeyInfo", "DeletionProtectionEnabled": false + })) + .unwrap(), + ) + .await + .unwrap(); + let before = queries.0.load(Ordering::Relaxed); + cached.table_key_info(account_id, "KeyInfo").await.unwrap(); + assert_eq!( + queries.0.load(Ordering::Relaxed), + before + 1, + "a local update invalidates backend metadata" + ); + cached + .delete_table( + account_id, + extenddb_core::types::DeleteTableInput { + table_name: "KeyInfo".into(), + }, + ) + .await + .unwrap(); + assert!(matches!( + cached.table_key_info(account_id, "KeyInfo").await, + Err(StorageError::TableNotFound(_)) + )); + cached.create_table(account_id, create()).await.unwrap(); + let next = cached.table_key_info(account_id, "KeyInfo").await.unwrap(); + assert_ne!(next.table_id, first.table_id); + + // A different backend represents a remote management writer. Its recreation + // is observed after expiry; cached metadata cannot alias the new generation. + fresh + .delete_table( + account_id, + extenddb_core::types::DeleteTableInput { + table_name: "KeyInfo".into(), + }, + ) + .await + .unwrap(); + fresh.create_table(account_id, create()).await.unwrap(); + let current = fresh.table_key_info(account_id, "KeyInfo").await.unwrap(); + assert_ne!(current.table_id, next.table_id); + let key = Item::from([("pk".into(), AttributeValue::S("same".into()))]); + assert!( + matches!( + cached.get_item(&next, &key).await, + Err(StorageError::TableNotFound(_)) + ), + "an old key-info generation must not read a recreated table" + ); + tokio::time::sleep(std::time::Duration::from_millis(550)).await; + assert_eq!( + cached + .table_key_info(account_id, "KeyInfo") + .await + .unwrap() + .table_id, + current.table_id + ); + host.shutdown().await.unwrap(); +} diff --git a/tests/elastic_cells/table_residency.rs b/tests/elastic_cells/table_residency.rs index 4402115..a3b8dc3 100644 --- a/tests/elastic_cells/table_residency.rs +++ b/tests/elastic_cells/table_residency.rs @@ -5,6 +5,158 @@ const ACCOUNT: &str = "123456789012"; // One account and two indexed tables, each with data, index and both directory roots. const CELL_CAPACITY: usize = 9; +struct ReleaseRetiredDirectory { + handle: cellule_runtime::cell::actor::CellHandle, + client: CellClient, + target: cellule_runtime::identity::CellTarget, + spec: beyonddb::DirectorySpec, + calls: Arc, +} + +impl cellule_runtime::client::LocalCellResolver for ReleaseRetiredDirectory { + fn resolve( + &self, + _: cellule_runtime::identity::CellTarget, + ) -> std::pin::Pin< + Box< + dyn std::future::Future< + Output = cellule_runtime::Result< + Option, + >, + > + Send, + >, + > { + let handle = self.handle.clone(); + let client = self.client.clone(); + let target = self.target.clone(); + let spec = self.spec.clone(); + let release = self.calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst) == 1; + Box::pin(async move { + if release { + // The first controller call discovers the pending root. Its + // next call follows the durable retirement receipt: release + // that exact root before either confirmation or acknowledgement. + let state = client + .query::(&target, None, Json(())) + .await + .unwrap() + .output + .0 + .unwrap(); + assert_eq!(state.spec, spec); + assert_eq!(state.mode, beyonddb::DirectoryMode::Retired); + handle.drain().await?; + } + Ok(None) + }) + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn deletion_acknowledges_retirement_after_directory_owner_release() { + let application = Arc::new( + Beyonddb::compile(BuildDescriptor { + source_revision: "retirement-receipt-gap".into(), + cargo_lock_digest: Digest::from_bytes([1; 32]), + }) + .unwrap(), + ); + let files = tempfile::tempdir().unwrap(); + let account = account_target(ACCOUNT).unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + object_store::path::Path::from("retirement-receipt-gap"), + *account.application().as_bytes(), + ); + let session = SessionId::from_bytes([199; 16]); + let host = CellNodeBuilder::new(application.clone()) + .with_runtime( + SqlWorkerPool::new(1, CELL_CAPACITY).unwrap(), + 16 * 1024 * 1024, + ) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(1 << 30))) + .with_session(session) + .build_unleased_for_maintenance() + .unwrap(); + let provisioner = Arc::new( + CellInitialPartitionProvisioner::new( + host.runtime(), + application.clone(), + layout.clone(), + session, + "https://retirement-gap.internal".into(), + files.path().into(), + ) + .unwrap(), + ); + provisioner.admit_account(ACCOUNT).await.unwrap(); + let client = CellClient::local_runtime(application.registry(), host.runtime(), layout.clone()); + let storage = + CellStorage::new(client.clone(), "us-east-1").with_initial_partitions(provisioner.clone()); + storage + .create_table(ACCOUNT, table("RetirementGap")) + .await + .unwrap(); + let record = client + .query::(&account, None, Json("RetirementGap".into())) + .await + .unwrap() + .output + .0 + .unwrap(); + let spec = beyonddb::DirectorySpec::root(record.id.clone()); + let target = beyonddb::directory_target(ACCOUNT, &spec).unwrap(); + let handle = provisioner + .admit_existing_directory(ACCOUNT, &spec) + .await + .unwrap(); + storage + .delete_table( + ACCOUNT, + DeleteTableInput { + table_name: "RetirementGap".into(), + }, + ) + .await + .unwrap(); + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let controlled = client + .clone() + .with_local_resolver(Arc::new(ReleaseRetiredDirectory { + client: CellClient::local(application.registry(), handle.clone()), + handle, + target: target.clone(), + spec, + calls: calls.clone(), + })); + let result = provisioner + .continue_table_deletion(&controlled, ACCOUNT, &record.id) + .await; + assert!(calls.load(std::sync::atomic::Ordering::SeqCst) >= 2); + let control = CellAuthority::new(layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(control.value().owner.is_none()); + assert!(control.value().root.is_some()); + // Even after release, the account must acknowledge this root and discover + // the independent index directory for the next bounded controller step. + let pending = client + .query::(&account, None, Json(record.id.clone())) + .await + .unwrap() + .output + .0 + .unwrap(); + host.shutdown().await.unwrap(); + assert!( + result.is_ok(), + "retired directory residency blocked acknowledgement: {result:?}" + ); + assert_eq!(pending.spec.table_id, record.global_secondary_indexes[0].id); +} + fn table(name: &str) -> extenddb_core::types::CreateTableInput { serde_json::from_value(serde_json::json!({ "TableName": name, diff --git a/tests/elastic_cells/transaction_commit.rs b/tests/elastic_cells/transaction_commit.rs new file mode 100644 index 0000000..6f7d559 --- /dev/null +++ b/tests/elastic_cells/transaction_commit.rs @@ -0,0 +1,189 @@ +use crate::*; +use beyonddb::{ + CommitPreparedTransaction, CommitPreparedTransactionInput, CommitPreparedTransactionOutcome, + CoordinatorPrepareReceipt, +}; + +fn mutation() -> MutationIdentity { + MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + ..identity(209) + } +} + +/// The surrounding driver fixture has recorded one of two actual prepares. +pub(super) async fn assert_atomic_prepared_commit( + client: &CellClient, + bootstrap: &Bootstrap<'_>, + coordinator_handle: &CellHandle, + application: Arc, + directory: &std::path::Path, + second: &(CellTarget, CoordinatorParticipant), + transaction_id: [u8; 16], +) -> std::path::PathBuf { + let account_id = "123456789012"; + let coordinator = coordinator_target(account_id, &transaction_id).unwrap(); + let transaction = ReadCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id, + routing_key: transaction_id.to_vec(), + }; + let input = |prepares| { + Json(CommitPreparedTransactionInput { + transaction: transaction.clone(), + prepares, + }) + }; + let partial = client + .command::(&coordinator, mutation(), input(vec![])) + .await; + assert!(matches!(partial, Err(InvocationError::Rejected(result)) + if result.output.0 == CommitPreparedTransactionOutcome::Decision( + DecideCrossCellTransactionOutcome::NotPrepared))); + for (position, participant_cell, expected) in [ + (1, [99; 32], CoordinatorPhaseOutcome::WrongParticipant), + ( + 2, + *second.0.cell_id().as_bytes(), + CoordinatorPhaseOutcome::Missing, + ), + ] { + let wrong = client + .command::( + &coordinator, + mutation(), + input(vec![CoordinatorPrepareReceipt { + position, + participant_cell, + sequence: 1, + }]), + ) + .await; + assert!(matches!(wrong, Err(InvocationError::Rejected(result)) + if result.output.0 == CommitPreparedTransactionOutcome::EvidenceRejected(vec![expected]))); + } + let invalid_sequence = client + .command::( + &coordinator, + mutation(), + input(vec![CoordinatorPrepareReceipt { + position: 1, + participant_cell: *second.0.cell_id().as_bytes(), + sequence: 0, + }]), + ) + .await; + assert!(invalid_sequence.is_err()); + let before = client + .query::(&coordinator, None, Json(transaction.clone())) + .await + .unwrap(); + let status = before.output.0.unwrap(); + assert_eq!(status.decision, CoordinatorDecision::Begin); + assert_eq!(status.prepared_count, 1); + assert_eq!(status.resolved_count, 0); + + let CoordinatorParticipantTarget::Data { + table_id, epoch, .. + } = &second.1.target + else { + panic!("fixture requires a data participant"); + }; + let prepared = transaction_command!( + client, + PreparePartitionTransaction, + &second.0, + mutation(), + Json(PreparePartitionTransactionInput { + table_id: table_id.clone(), + epoch: *epoch, + transaction_id, + coordinator_cell: *coordinator.cell_id().as_bytes(), + coordinator_key: transaction_id.to_vec(), + operations: second + .1 + .operations + .iter() + .map(|op| op.operation.clone()) + .collect(), + }), + ) + .await + .unwrap(); + let commit_input = input(vec![CoordinatorPrepareReceipt { + position: 1, + participant_cell: *second.0.cell_id().as_bytes(), + sequence: prepared.receipt.commit_sequence, + }]); + let commit_identity = mutation(); + let committed = client + .command::(&coordinator, commit_identity, commit_input.clone()) + .await + .unwrap(); + assert_eq!( + committed.output.0, + CommitPreparedTransactionOutcome::Decision(DecideCrossCellTransactionOutcome::Decided( + CoordinatorDecision::Commit + )) + ); + assert_eq!( + committed.receipt.commit_sequence - before.receipt.commit_sequence, + 1, + "prepare evidence and terminal decision require one durable coordinator commit" + ); + let replay = client + .command::(&coordinator, commit_identity, commit_input) + .await + .unwrap(); + assert_eq!( + replay.receipt.commit_sequence, + committed.receipt.commit_sequence + ); + assert_eq!(replay.output.0, committed.output.0); + + // Restore the coordinator from its durable root before the driver resolves + // participants. The decision and both prepare receipts must survive together. + coordinator_handle.drain().await.unwrap(); + let provisioner = CellInitialPartitionProvisioner::new( + bootstrap.runtime.clone(), + application, + bootstrap.layout.clone(), + bootstrap.session, + "https://commit-restart.internal:8081".into(), + directory.join("commit-restored"), + ) + .unwrap(); + provisioner + .admit_coordinator(account_id, &transaction_id) + .await + .unwrap(); + let restored = client + .query::(&coordinator, None, Json(transaction)) + .await + .unwrap() + .output + .0 + .unwrap(); + assert_eq!(restored.decision, CoordinatorDecision::Commit); + assert_eq!(restored.prepared_count, 2); + assert_eq!(restored.resolved_count, 0); + let restored_directory = directory.join("commit-restored").join( + blake3::Hash::from_bytes(*coordinator.cell_id().as_bytes()) + .to_hex() + .as_str(), + ); + let files = std::fs::read_dir(restored_directory) + .unwrap() + .map(|entry| entry.unwrap().path()) + .filter(|path| { + path.extension() + .is_some_and(|extension| extension == "sqlite") + }) + .collect::>(); + assert_eq!( + files.len(), + 1, + "one restored coordinator activation is expected" + ); + files[0].clone() +} diff --git a/tests/elastic_cells/transaction_driver.rs b/tests/elastic_cells/transaction_driver.rs index 37031b7..c83a59a 100644 --- a/tests/elastic_cells/transaction_driver.rs +++ b/tests/elastic_cells/transaction_driver.rs @@ -1,5 +1,5 @@ use crate::*; -use beyonddb::{TableRecord, TransactionFailure}; +use beyonddb::{ReadCoordinatorResume, TableRecord, TransactionFailure}; #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures() { @@ -119,14 +119,15 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures for scenario in [180_u8, 181, 182, 183, 184, 185, 186, 187, 188] { let transaction_id = [scenario; 16]; let coordinator = coordinator_target(account_id, &transaction_id).unwrap(); + let mut coordinator_file = directory + .path() + .join(format!("coordinator-{scenario}.sqlite")); let coordinator_handle = bootstrap .cell( &coordinator, "beyonddb-coordinator", scenario, - &directory - .path() - .join(format!("coordinator-{scenario}.sqlite")), + &coordinator_file, initialize_coordinator, ) .await; @@ -186,7 +187,40 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures ) .await .unwrap(); - let mut recorded_sequence = None; + let snapshot = client + .query::( + &coordinator, + None, + Json(ReadCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id, + routing_key: transaction_id.to_vec(), + }), + ) + .await + .unwrap() + .output + .0; + assert_eq!( + snapshot.status.unwrap().decision, + CoordinatorDecision::Begin + ); + if scenario == 188 { + assert!( + snapshot.participants.is_none(), + "large escaped payloads must retain chunked retrieval" + ); + } else { + assert_eq!( + snapshot + .participants + .unwrap() + .into_iter() + .map(|p| p.participant) + .collect::>(), + request + ); + } if matches!(scenario, 180 | 181 | 183) { // Cover recorded and lost prepare receipts, plus a conflicting // transaction on the second participant. @@ -217,25 +251,50 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures .await .unwrap(); if scenario == 180 { - recorded_sequence = Some( - client - .command::( - &coordinator, - identity(179), - Json(CoordinatorPhaseInput { - account_id: account_id.into(), - transaction_id, - routing_key: transaction_id.to_vec(), - position: 0, - participant_cell: *participants[0].0.cell_id().as_bytes(), - sequence: prepared.receipt.commit_sequence, - }), - ) - .await - .unwrap() - .receipt - .commit_sequence, - ); + client + .command::( + &coordinator, + identity(179), + Json(CoordinatorPhaseInput { + account_id: account_id.into(), + transaction_id, + routing_key: transaction_id.to_vec(), + position: 0, + participant_cell: *participants[0].0.cell_id().as_bytes(), + sequence: prepared.receipt.commit_sequence, + }), + ) + .await + .unwrap(); + let partial = client + .query::( + &coordinator, + None, + Json(ReadCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id, + routing_key: transaction_id.to_vec(), + }), + ) + .await + .unwrap() + .output + .0; + assert_eq!(partial.status.unwrap().prepared_count, 1); + let remaining = partial.participants.unwrap(); + assert_eq!(remaining.len(), 1); + assert_eq!(remaining[0].position, 1); + assert_eq!(remaining[0].participant, request[1]); + coordinator_file = super::transaction_commit::assert_atomic_prepared_commit( + &client, + &bootstrap, + &coordinator_handle, + application.clone(), + directory.path(), + &participants[1], + transaction_id, + ) + .await; } } let decision = if scenario == 181 { @@ -289,7 +348,9 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures } else { Arc::new(RefusePhase { inner: transport, - command: 12, + // These scenarios send small participant payloads, so the + // production adapter selects the bounded prepare opcode. + command: 28, persistent: scenario == 187, winner: (scenario == 186).then(|| (client.clone(), transaction_id)), refused: lost.clone(), @@ -370,18 +431,10 @@ async fn driver_resumes_prepares_and_resolves_commit_condition_and_lock_failures ) .await .unwrap(); - if let Some(sequence) = recorded_sequence { - // The remaining prepare and decision need two commits. Nearby - // resolutions can share one commit; the progress timer may split - // them when a participant finishes later. - assert!((3..=4).contains(&(status.receipt.commit_sequence - sequence))); - } let status = status.output.0.unwrap(); assert_eq!(status.resolved_count, 2); let database = rusqlite::Connection::open_with_flags( - directory - .path() - .join(format!("coordinator-{scenario}.sqlite")), + &coordinator_file, rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY, ) .unwrap(); @@ -559,11 +612,14 @@ impl PeerRoundTrip for DropPhaseReplies { Some(peer_request::Operation::Mutate(mutation)) => match &mutation.operation { Some(mutation_request::Operation::CellCommand(command)) => { match command.command_id { - 12 | 21 => 1, - 3 => 2, + 12 | 21 | 28 | 57 => 1, + 3 | 11 => 2, 1 => 4, 13 | 22 => 8, 5 => 16, + // Separate bit lets fresh completion tests lose the + // batched coordinator receipt without losing BEGIN. + 10 => 128, 14 => 32, 23 => 64, _ => 0, diff --git a/tests/elastic_cells/transaction_reads.rs b/tests/elastic_cells/transaction_reads.rs index 5d1d7f1..f09349f 100644 --- a/tests/elastic_cells/transaction_reads.rs +++ b/tests/elastic_cells/transaction_reads.rs @@ -1,7 +1,8 @@ use crate::*; use beyonddb::{ - ReadAccountTransactionResult, ReadPartitionTransactionResult, ReadTransactionResultInput, - TransactionReadResult, + BoundedTransactionReadResult, ReadAccountTransactionResult, + ReadAccountTransactionResultBounded, ReadPartitionTransactionResult, + ReadPartitionTransactionResultBounded, ReadTransactionResultInput, TransactionReadResult, }; use extenddb_core::types::TableKeyInfo; @@ -258,7 +259,23 @@ async fn saved( }, position: 0, }); - match participant { + let bounded = match participant { + CoordinatorParticipantTarget::Account => { + client + .query::(target, None, input.clone()) + .await + .unwrap() + .output + } + CoordinatorParticipantTarget::Data { .. } => { + client + .query::(target, None, input.clone()) + .await + .unwrap() + .output + } + }; + let wide = match participant { CoordinatorParticipantTarget::Account => { client .query::(target, None, input) @@ -275,5 +292,17 @@ async fn saved( .output .0 } + }; + match bounded { + BoundedTransactionReadResult::Unavailable => { + assert_eq!(wide, TransactionReadResult::Unavailable) + } + BoundedTransactionReadResult::Item(item) => { + assert_eq!(wide, TransactionReadResult::Item(item)) + } + BoundedTransactionReadResult::WideRequired => { + assert!(matches!(wide, TransactionReadResult::Item(Some(_)))) + } } + wide } diff --git a/tests/elastic_cells/transaction_resume.rs b/tests/elastic_cells/transaction_resume.rs new file mode 100644 index 0000000..7a1dbba --- /dev/null +++ b/tests/elastic_cells/transaction_resume.rs @@ -0,0 +1,267 @@ +use crate::*; +use beyonddb::{ReadCoordinatorResume, TableRecord}; +use cellule_runtime::fleet::telemetry::{ + CellTelemetry, PrimitiveOperationKind, PrimitiveOperationOutcome, +}; + +#[derive(Default)] +struct CoordinatorQueries(std::sync::atomic::AtomicU64); +impl CellTelemetry for CoordinatorQueries { + fn primitive_operation( + &self, + module: &'static str, + kind: PrimitiveOperationKind, + _: PrimitiveOperationOutcome, + _: std::time::Duration, + ) { + if module == "beyonddb-coordinator" && kind == PrimitiveOperationKind::Query { + self.0.fetch_add(1, Ordering::Relaxed); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn small_transaction_resume_uses_one_coordinator_payload_snapshot() { + let application = Arc::new( + Beyonddb::compile(BuildDescriptor { + source_revision: "transaction-resume".into(), + cargo_lock_digest: Digest::from_bytes([1; 32]), + }) + .unwrap(), + ); + let account_id = "123456789012"; + let account = account_target(account_id).unwrap(); + let session = SessionId::from_bytes([180; 16]); + let directory = tempfile::TempDir::new().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + object_store::path::Path::from("transaction-resume"), + *account.application().as_bytes(), + ); + let host = CellNodeBuilder::new(application.clone()) + .with_runtime(SqlWorkerPool::new(1, 17).unwrap(), 16 * 1024 * 1024) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(1 << 30))) + .with_session(session) + .build_unleased_for_maintenance() + .unwrap(); + let queries = Arc::new(CoordinatorQueries::default()); + host.install_telemetry(queries.clone()).unwrap(); + let registry = application.registry(); + let bootstrap = Bootstrap { + runtime: host.runtime(), + registry: ®istry, + layout: &layout, + session, + }; + let mut table_bytes = [180; 32]; + table_bytes[..16].copy_from_slice(account.tenant().as_bytes()); + let table = TableRecord { + table_class: Default::default(), + table_class_updates_ms: Vec::new(), + placement: beyonddb::TablePlacement::Routed { + initial_partitions: 2, + }, + local_secondary_indexes: Vec::new(), + global_secondary_indexes: Vec::new(), + id: blake3::Hash::from_bytes(table_bytes).to_hex().to_string(), + created_at_ms: 1000, + table_name: "Driver".into(), + key_schema: vec![KeySchemaElement { + attribute_name: "id".into(), + key_type: KeyType::Hash, + }], + attribute_definitions: vec![AttributeDefinition { + attribute_name: "id".into(), + attribute_type: ScalarAttributeType::S, + }], + billing_mode: BillingMode::PayPerRequest, + provisioned_throughput: None, + deletion_protection_enabled: false, + pay_per_request_since_ms: Some(1000), + stream: None, + }; + let client = CellClient::local_runtime(registry.clone(), host.runtime(), layout.clone()); + let boundary = [0x80, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]; + let mut participants = Vec::new(); + let mut handles = Vec::new(); + for (position, left) in [true, false].into_iter().enumerate() { + let partition_id = [u8::try_from(position + 1).unwrap(); 16]; + let target = data_target(account_id, &table.id, &partition_id).unwrap(); + let handle = bootstrap + .cell( + &target, + "beyonddb-data", + partition_id[0], + &directory.path().join(format!("data-{position}.sqlite")), + initialize_partition, + ) + .await; + handles.push(handle); + client + .command::( + &target, + identity(180), + Json(PartitionInstall::Serving(PartitionSpec { + table: table.clone(), + partition_id, + lower: (!left).then_some(boundary), + upper: left.then_some(boundary), + epoch: 1, + })), + ) + .await + .unwrap(); + let mut item = key_in_range(&table.id, &table.key_schema, left, 1000); + item.insert("value".into(), AttributeValue::N("1".into())); + participants.push(( + target, + CoordinatorParticipant { + target: CoordinatorParticipantTarget::Data { + table_id: table.id.clone(), + partition_id, + epoch: 1, + }, + operations: vec![IndexedTransactionOperation { + index: u8::try_from(position).unwrap(), + operation: TransactionOperation::Put(PutItemInput { + table_name: table.table_name.clone(), + table_id: table.id.clone(), + item, + condition: None, + }), + }], + }, + )); + } + participants.sort_by_key(|(target, _)| *target.cell_id().as_bytes()); + // Operation indexes deliberately oppose participant order. + for (position, (_, participant)) in participants.iter_mut().enumerate() { + participant.operations[0].index = u8::try_from(1 - position).unwrap(); + } + + let transaction_id = [199; 16]; + let coordinator = coordinator_target(account_id, &transaction_id).unwrap(); + let coordinator_handle = bootstrap + .cell( + &coordinator, + "beyonddb-coordinator", + 199, + &directory.path().join("coordinator.sqlite"), + initialize_coordinator, + ) + .await; + handles.push(coordinator_handle); + let client = + CellClient::local_many_with_telemetry(registry, handles, host.runtime().telemetry_handle()) + .unwrap(); + let storage = CellStorage::new(client.clone(), "us-east-1"); + transaction_command!( + client, + BeginCrossCellTransaction, + &coordinator, + identity(199), + Json(BeginCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id, + token: None, + participants: participants.iter().map(|(_, p)| p.clone()).collect() + }) + ) + .await + .unwrap(); + let read = ReadCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id, + routing_key: transaction_id.to_vec(), + }; + let initial = client + .query::(&coordinator, None, Json(read.clone())) + .await + .unwrap() + .output + .0; + assert_eq!( + initial.status.as_ref().unwrap().decision, + CoordinatorDecision::Begin + ); + assert_eq!( + initial + .participants + .as_ref() + .unwrap() + .iter() + .map(|p| p.participant.clone()) + .collect::>(), + participants + .iter() + .map(|(_, p)| p.clone()) + .collect::>() + ); + let missing = client + .query::( + &coordinator, + None, + Json(ReadCrossCellTransactionInput { + transaction_id: [198; 16], + ..read.clone() + }), + ) + .await + .unwrap() + .output + .0; + assert!(missing.status.is_none() && missing.participants.is_none()); + assert!( + client + .query::( + &coordinator, + None, + Json(ReadCrossCellTransactionInput { + account_id: "999999999999".into(), + ..read.clone() + }) + ) + .await + .is_err() + ); + let before = queries.0.load(Ordering::Relaxed); + let decision = storage + .resume_cross_cell_transaction(account_id, &transaction_id, transaction_id) + .await + .unwrap(); + assert_eq!(decision, CoordinatorDecision::Commit); + let reads = queries.0.load(Ordering::Relaxed) - before; + println!("coordinator queries for a two-participant resume: {reads}"); + assert_eq!( + reads, 4, + "small resume must fetch status and immutable payloads in one query, then retain decision and resolution checks" + ); + let status = client + .query::( + &coordinator, + None, + Json(ReadCrossCellTransactionInput { + account_id: account_id.into(), + transaction_id, + routing_key: transaction_id.to_vec(), + }), + ) + .await + .unwrap() + .output + .0 + .unwrap(); + assert_eq!(status.resolved_count, 2); + let terminal = client + .query::(&coordinator, None, Json(read)) + .await + .unwrap() + .output + .0; + assert_eq!(terminal.status.unwrap(), status); + assert!( + terminal.participants.is_none(), + "compacted or terminal transactions must not reuse payloads" + ); + host.runtime().shutdown().await.unwrap(); +} diff --git a/tests/node_log_authority.rs b/tests/node_log_authority.rs index 0d1afa7..146b341 100644 --- a/tests/node_log_authority.rs +++ b/tests/node_log_authority.rs @@ -17,6 +17,12 @@ use ed25519_dalek::SigningKey; use object_store::{memory::InMemory, path::Path}; use tokio_util::sync::CancellationToken; +#[path = "node_log_authority/stalled_log_read.rs"] +mod stalled_log_read; + +#[path = "node_log_authority/heartbeat_versions.rs"] +mod heartbeat_versions; + const FLEET: Digest = Digest::from_bytes([80; 32]); const IMAGE: Digest = Digest::from_bytes([81; 32]); const RELEASE: Digest = Digest::from_bytes([82; 32]); @@ -121,6 +127,8 @@ async fn published_log_authority_reconciles_heartbeat_and_fences_transitions() { authority.recruit(1, 4096, 16).await.unwrap(), Some(vec![follower_node]) ); + assert!(!authority.rotation_required(1, 16).await.unwrap()); + assert!(authority.rotation_required(2, 16).await.is_err()); assert!(authority.activate(2).await.is_err()); let cancellation = CancellationToken::new(); @@ -192,4 +200,89 @@ async fn published_log_authority_reconciles_heartbeat_and_fences_transitions() { authority.activate(1).await, Err(cellule_runtime::Error::Fenced) )); + assert!(matches!( + authority.rotation_required(1, 16).await, + Err(cellule_runtime::Error::Fenced) + )); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn current_log_membership_requires_rotation_after_follower_expiry() { + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("beyonddb-node-log-member-expiry"), + [44; 16], + ); + let directory = NodeDirectory::new(layout, FLEET, IMAGE, RELEASE); + let leader_node = NodeId::from_bytes([101; 16]); + let leader_session = SessionId::from_bytes([102; 16]); + let follower_node = NodeId::from_bytes([103; 16]); + let follower_session = SessionId::from_bytes([104; 16]); + let now = now_ms(); + let follower_expires = now + 5_000; + directory + .create( + advertisement( + follower_node, + follower_session, + 105, + NodeCapacity { + free_memory_bytes: 16 * 1024 * 1024, + free_disk_bytes: 1 << 30, + follower_free_bytes: 1 << 30, + job_credits: 8, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + ..NodeCapacity::default() + }, + now, + follower_expires, + ) + .unwrap(), + now, + ) + .await + .unwrap(); + let published = NodeLeasePublisher::new(directory.clone(), move |now, expires| { + advertisement( + leader_node, + leader_session, + 106, + NodeCapacity { + free_memory_bytes: 16 * 1024 * 1024, + free_disk_bytes: 1 << 30, + job_credits: 8, + ..NodeCapacity::default() + }, + now, + expires, + ) + }) + .publish() + .await + .unwrap(); + let authority = published.log_authority(); + assert_eq!( + authority.recruit(1, 4096, 16).await.unwrap(), + Some(vec![follower_node]) + ); + assert!(!authority.rotation_required(1, 16).await.unwrap()); + // A different epoch is an error, rather than a healthy membership result. + assert!(authority.rotation_required(2, 16).await.is_err()); + tokio::time::timeout(Duration::from_secs(7), async { + while now_ms() <= follower_expires { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + assert!(authority.rotation_required(1, 16).await.unwrap()); + // Liveness observation cannot silently replace the enrolled member set. + let current = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + let log = current.advertisement().log().unwrap(); + assert_eq!(log.epoch(), 1); + assert_eq!(log.members(), &[follower_node]); } diff --git a/tests/node_log_authority/heartbeat_versions.rs b/tests/node_log_authority/heartbeat_versions.rs new file mode 100644 index 0000000..58f1fdb --- /dev/null +++ b/tests/node_log_authority/heartbeat_versions.rs @@ -0,0 +1,309 @@ +use super::*; +use cellule_store::{StorageObservation, StorageObserver, StorageOperation, StorageOutcome}; +use std::sync::Mutex; + +#[derive(Default)] +struct RefreshTrace { + // Arm only after setup. No SDK or background requests share this store. + active: Mutex)>>, +} + +impl RefreshTrace { + fn arm(&self) -> CancellationToken { + let done = CancellationToken::new(); + *self.active.lock().unwrap() = Some((done.clone(), Vec::new())); + done + } + + fn take(&self) -> Vec { + self.active.lock().unwrap().take().unwrap().1 + } +} + +impl StorageObserver for RefreshTrace { + fn started(&self, _operation: StorageOperation) {} + + fn finished(&self, observation: StorageObservation) { + if let Some((done, observations)) = self.active.lock().unwrap().as_mut() { + // Freeze at this publication's successful PUT. A missed interval + // can start the next heartbeat before the caller cancels the task. + if done.is_cancelled() { + return; + } + observations.push(observation); + if observation.operation == StorageOperation::Put + && observation.outcome == StorageOutcome::Success + { + done.cancel(); + } + } + } +} + +async fn refresh_after_coverage(external_update: bool, slow_refresh: bool) { + use object_store::throttle::{ThrottleConfig, ThrottledStore}; + + let trace = Arc::new(RefreshTrace::default()); + let store = Arc::new(ThrottledStore::new( + InMemory::new(), + ThrottleConfig::default(), + )); + let layout = CellStorageLayout::new( + Store::new(store.clone()).with_storage_observer(trace.clone()), + Path::from("heartbeat-after-coverage"), + [49; 16], + ); + let directory = NodeDirectory::new(layout, FLEET, IMAGE, RELEASE); + let leader_node = NodeId::from_bytes([110; 16]); + let leader_session = SessionId::from_bytes([111; 16]); + let follower_node = NodeId::from_bytes([112; 16]); + let now = now_ms(); + directory + .create( + advertisement( + follower_node, + SessionId::from_bytes([113; 16]), + 114, + NodeCapacity { + free_memory_bytes: 16 << 20, + free_disk_bytes: 1 << 30, + follower_free_bytes: 1 << 30, + job_credits: 8, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + ..NodeCapacity::default() + }, + now, + now + 15_000, + ) + .unwrap(), + now, + ) + .await + .unwrap(); + let published = NodeLeasePublisher::new(directory.clone(), move |now, expires| { + advertisement( + leader_node, + leader_session, + 115, + NodeCapacity::default(), + now, + expires, + ) + }) + .publish() + .await + .unwrap(); + let guard = published.guard(); + let authority = published.log_authority(); + authority.recruit(1, 4096, 16).await.unwrap(); + authority.activate(1).await.unwrap(); + authority.advance_coverage(1, 1).await.unwrap(); + let mut before = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + if external_update { + // Bypass the publisher's adapter: its ETag hint must not become authority. + before = directory + .advance_log_coverage(&before, 2, now_ms()) + .await + .unwrap(); + } + if slow_refresh { + // One 6.5-second CAS leaves room for repeated renewals. A stale CAS, + // reload and second CAS exceed the first lease's 12-second remainder. + store.config_mut(|config| config.wait_put_per_call = Duration::from_millis(6_500)); + } + let done = trace.arm(); + let cancellation = CancellationToken::new(); + let run_cancellation = cancellation.clone(); + let heartbeat = tokio::spawn(async move { published.run(&run_cancellation).await }); + let mut completed = tokio::time::timeout(Duration::from_secs(18), async { + tokio::select! { + () = done.cancelled() => true, + () = guard.wait_fenced() => false, + } + }) + .await; + let mut observations = Vec::new(); + if slow_refresh && matches!(completed, Ok(true)) { + observations.extend(trace.take()); + let second = trace.arm(); + completed = tokio::time::timeout(Duration::from_secs(18), async { + tokio::select! { + () = second.cancelled() => true, + () = guard.wait_fenced() => false, + } + }) + .await; + } + cancellation.cancel(); + let stopped = tokio::time::timeout(Duration::from_secs(2), heartbeat).await; + observations.extend(trace.take()); + let puts = observations + .iter() + .filter(|o| o.operation == StorageOperation::Put) + .count(); + let gets = observations + .iter() + .filter(|o| o.operation == StorageOperation::Get) + .count(); + let conflicts = observations + .iter() + .filter(|o| o.outcome == StorageOutcome::Conflict) + .count(); + eprintln!( + "heartbeat refresh after coverage external={external_update}, slow={slow_refresh}: puts={puts}, gets={gets}, conflicts={conflicts}, renewed={completed:?}" + ); + let stopped = stopped.unwrap().unwrap(); + assert!( + completed.unwrap(), + "avoidable stale CAS exhausted the serving lease: {stopped:?}" + ); + stopped.unwrap(); + let current = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + assert!(current.advertisement().generation() > before.advertisement().generation()); + assert!(current.advertisement().expires_at_ms() > before.advertisement().expires_at_ms()); + let log = current.advertisement().log().unwrap(); + assert!(log.active()); + assert_eq!(log.members(), &[follower_node]); + assert_eq!(log.tiered_through(), if external_update { 2 } else { 1 }); + guard.check().unwrap(); + guard.fence(); + if external_update { + assert_eq!((puts, gets, conflicts), (2, 1, 1)); + } else { + assert_eq!( + (puts, gets, conflicts), + (if slow_refresh { 2 } else { 1 }, 0, 0), + "known log coverage must not force a stale heartbeat CAS and reload" + ); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn heartbeat_uses_completed_local_log_version_without_redundant_io() { + refresh_after_coverage(false, false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn heartbeat_rebases_unseen_log_version_through_authoritative_cas() { + refresh_after_coverage(true, false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn heartbeat_survives_slow_cas_after_completed_log_coverage() { + refresh_after_coverage(false, true).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn delayed_authority_read_cannot_regress_the_next_heartbeat_etag() { + let trace = Arc::new(RefreshTrace::default()); + let store = Arc::new(super::stalled_log_read::HeldReadStore::default()); + let layout = CellStorageLayout::new( + Store::new(store.clone()).with_storage_observer(trace.clone()), + Path::from("heartbeat-late-observation"), + [49; 16], + ); + let directory = NodeDirectory::new(layout.clone(), FLEET, IMAGE, RELEASE); + let leader_node = NodeId::from_bytes([110; 16]); + let leader_session = SessionId::from_bytes([111; 16]); + let follower_node = NodeId::from_bytes([112; 16]); + let now = now_ms(); + directory + .create( + advertisement( + follower_node, + SessionId::from_bytes([113; 16]), + 114, + NodeCapacity { + free_memory_bytes: 16 << 20, + free_disk_bytes: 1 << 30, + follower_free_bytes: 1 << 30, + job_credits: 8, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + ..NodeCapacity::default() + }, + now, + now + 15_000, + ) + .unwrap(), + now, + ) + .await + .unwrap(); + let published = NodeLeasePublisher::new(directory.clone(), move |now, expires| { + advertisement( + leader_node, + leader_session, + 115, + NodeCapacity::default(), + now, + expires, + ) + }) + .publish() + .await + .unwrap(); + let guard = published.guard(); + let authority = published.log_authority(); + authority.recruit(1, 4096, 16).await.unwrap(); + authority.activate(1).await.unwrap(); + authority.advance_coverage(1, 1).await.unwrap(); + + *store.held_path.lock().unwrap() = Some(layout.node_path(leader_session.as_bytes())); + let check = authority.clone(); + let rotation = tokio::spawn(async move { check.rotation_required(1, 16).await }); + let entered = tokio::time::timeout(Duration::from_secs(2), store.entered.cancelled()).await; + let first = trace.arm(); + let cancellation = CancellationToken::new(); + let run_cancellation = cancellation.clone(); + let heartbeat = tokio::spawn(async move { published.run(&run_cancellation).await }); + let refreshed = tokio::time::timeout(Duration::from_secs(6), first.cancelled()).await; + // The CAS observer fires before the refresh future returns. Give the + // publisher its next poll before delivering the captured older response. + tokio::time::sleep(Duration::from_millis(50)).await; + trace.take(); + store.release.cancel(); + let checked = tokio::time::timeout(Duration::from_secs(2), rotation).await; + let second = trace.arm(); + let completed = tokio::time::timeout(Duration::from_secs(6), second.cancelled()).await; + cancellation.cancel(); + let stopped = tokio::time::timeout(Duration::from_secs(2), heartbeat).await; + let observations = trace.take(); + entered.unwrap(); + refreshed.unwrap(); + assert!(!checked.unwrap().unwrap().unwrap()); + completed.unwrap(); + stopped.unwrap().unwrap().unwrap(); + let puts = observations + .iter() + .filter(|o| o.operation == StorageOperation::Put) + .count(); + let gets = observations + .iter() + .filter(|o| o.operation == StorageOperation::Get) + .count(); + let conflicts = observations + .iter() + .filter(|o| o.outcome == StorageOutcome::Conflict) + .count(); + eprintln!( + "heartbeat after delayed authority read: puts={puts}, gets={gets}, conflicts={conflicts}" + ); + assert_eq!((puts, gets, conflicts), (1, 0, 0)); + let current = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.advertisement().log().unwrap().tiered_through(), 1); + assert!(current.advertisement().log().unwrap().active()); + guard.check().unwrap(); + guard.fence(); +} diff --git a/tests/node_log_authority/stalled_log_read.rs b/tests/node_log_authority/stalled_log_read.rs new file mode 100644 index 0000000..6eab138 --- /dev/null +++ b/tests/node_log_authority/stalled_log_read.rs @@ -0,0 +1,194 @@ +use super::*; +use futures_util::stream::BoxStream; +use object_store::{ + CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, + PutMultipartOptions, PutOptions, PutPayload, PutResult, +}; +use std::{fmt, sync::Mutex}; + +#[derive(Debug, Default)] +pub(super) struct HeldReadStore { + inner: InMemory, + pub(super) held_path: Mutex>, + pub(super) entered: CancellationToken, + pub(super) release: CancellationToken, +} + +impl fmt::Display for HeldReadStore { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str("HeldReadStore") + } +} + +#[async_trait::async_trait] +impl ObjectStore for HeldReadStore { + async fn put_opts( + &self, + path: &Path, + payload: PutPayload, + options: PutOptions, + ) -> object_store::Result { + self.inner.put_opts(path, payload, options).await + } + + async fn put_multipart_opts( + &self, + path: &Path, + options: PutMultipartOptions, + ) -> object_store::Result> { + self.inner.put_multipart_opts(path, options).await + } + + async fn get_opts(&self, path: &Path, options: GetOptions) -> object_store::Result { + let hold = { + let mut held = self.held_path.lock().unwrap(); + if held.as_ref() == Some(path) { + held.take(); + true + } else { + false + } + }; + // Capture this exact record and ETag before holding its response. A + // heartbeat can then publish a newer version while the log caller waits. + let result = self.inner.get_opts(path, options).await?; + if hold { + self.entered.cancel(); + self.release.cancelled().await; + } + Ok(result) + } + + fn delete_stream( + &self, + paths: BoxStream<'static, object_store::Result>, + ) -> BoxStream<'static, object_store::Result> { + self.inner.delete_stream(paths) + } + + fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { + self.inner.list(prefix) + } + + async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { + self.inner.list_with_delimiter(prefix).await + } + + async fn copy_opts( + &self, + from: &Path, + to: &Path, + options: CopyOptions, + ) -> object_store::Result<()> { + self.inner.copy_opts(from, to, options).await + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn stalled_log_read_does_not_block_heartbeat_and_rebases_coverage() { + let store = Arc::new(HeldReadStore::default()); + let layout = CellStorageLayout::new( + Store::new(store.clone()), + Path::from("heartbeat-during-log-read"), + [49; 16], + ); + let directory = NodeDirectory::new(layout.clone(), FLEET, IMAGE, RELEASE); + let leader_node = NodeId::from_bytes([110; 16]); + let leader_session = SessionId::from_bytes([111; 16]); + let follower_node = NodeId::from_bytes([112; 16]); + let now = now_ms(); + directory + .create( + advertisement( + follower_node, + SessionId::from_bytes([113; 16]), + 114, + NodeCapacity { + free_memory_bytes: 16 << 20, + free_disk_bytes: 1 << 30, + follower_free_bytes: 1 << 30, + job_credits: 8, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + ..NodeCapacity::default() + }, + now, + now + 15_000, + ) + .unwrap(), + now, + ) + .await + .unwrap(); + let published = NodeLeasePublisher::new(directory.clone(), move |now, expires| { + advertisement( + leader_node, + leader_session, + 115, + NodeCapacity::default(), + now, + expires, + ) + }) + .publish() + .await + .unwrap(); + let guard = published.guard(); + let authority = published.log_authority(); + authority.recruit(1, 4096, 16).await.unwrap(); + authority.activate(1).await.unwrap(); + let before = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + *store.held_path.lock().unwrap() = Some(layout.node_path(leader_session.as_bytes())); + let log_authority = authority.clone(); + let coverage = tokio::spawn(async move { log_authority.advance_coverage(1, 1).await }); + tokio::time::timeout(Duration::from_secs(2), store.entered.cancelled()) + .await + .unwrap(); + let cancellation = CancellationToken::new(); + let run_cancellation = cancellation.clone(); + let heartbeat = tokio::spawn(async move { published.run(&run_cancellation).await }); + let renewed = tokio::time::timeout(Duration::from_secs(6), async { + loop { + let current = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + if current.advertisement().expires_at_ms() > before.advertisement().expires_at_ms() { + return current; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await; + // Always release and join both tasks before asserting, including on the + // old implementation where the held log read prevents renewal entirely. + store.release.cancel(); + let covered = tokio::time::timeout(Duration::from_secs(3), coverage).await; + cancellation.cancel(); + let stopped = tokio::time::timeout(Duration::from_secs(2), heartbeat).await; + let renewed = renewed.expect("log storage I/O blocked the serving heartbeat"); + covered.unwrap().unwrap().unwrap(); + stopped.unwrap().unwrap().unwrap(); + let current = directory + .load(leader_session, now_ms()) + .await + .unwrap() + .unwrap(); + assert!(current.advertisement().expires_at_ms() >= renewed.advertisement().expires_at_ms()); + assert!(current.advertisement().generation() > renewed.advertisement().generation()); + let log = current.advertisement().log().unwrap(); + assert_eq!(log.epoch(), 1); + assert_eq!(log.members(), &[follower_node]); + assert!(log.active()); + assert_eq!(log.tiered_through(), 1); + guard.check().unwrap(); + guard.fence(); + assert!(matches!( + authority.advance_coverage(1, 2).await, + Err(cellule_runtime::Error::Fenced) + )); +} diff --git a/tests/peer_network.rs b/tests/peer_network.rs index 0176850..976e158 100644 --- a/tests/peer_network.rs +++ b/tests/peer_network.rs @@ -954,6 +954,31 @@ async fn run_signed_sdk_network_recovery() { }; let (restart_coordinator, _) = recovery::abandon_commit(&provisioner, &client, restart_id, "changed-endpoint").await; + // Retain the published range identity/progress before losing this owner. + // Capacity reclamation may already have left a range Idle on either node. + let mut range_baselines = HashMap::new(); + for name in ["NetworkData", "RemoteTable"] { + let table = client + .query::(&account, None, Json(name.into())) + .await + .unwrap() + .output + .0 + .unwrap(); + let route = crate::single_leaf_route(&client, &account, &table.id) + .await + .unwrap(); + for partition in route.partitions { + let target = + beyonddb::data_target("123456789012", &table.id, &partition.partition_id).unwrap(); + let control = CellAuthority::new(layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + range_baselines.insert(target.cell_id(), control.value().clone()); + } + } shutdown_tx.send(()).unwrap(); server.await.unwrap().unwrap(); owner_lease.cancel(); @@ -1049,9 +1074,11 @@ async fn run_signed_sdk_network_recovery() { .await .unwrap() .unwrap(); - assert_eq!( - current.value().owner.as_ref().unwrap().session, - replacement_session + recovery::assert_retained_range( + &range_baselines[&target.cell_id()], + current.value(), + replacement_session, + remote_session, ); } let table = replacement_client @@ -1074,9 +1101,11 @@ async fn run_signed_sdk_network_recovery() { .await .unwrap() .unwrap(); - assert_eq!( - authority.value().owner.as_ref().unwrap().session, - remote_session + recovery::assert_retained_range( + &range_baselines[&target.cell_id()], + authority.value(), + replacement_session, + remote_session, ); } // A live peer may already own a credential shard. Only the failed owner's @@ -1280,95 +1309,26 @@ async fn run_signed_sdk_network_recovery() { index_recovery .assert_settled(&replacement_sdk, &replacement_client, "before") .await; - index_recovery + let index_fences = index_recovery .assert_owner(&CellAuthority::new(layout.clone()), remote_session) .await; - // Install before this shard exists. Discovery must see later registrations - // without taking a live owner, then recover after that owner stops renewing. - replacement_provisioner - .install_transaction_recovery_loop( - &replacement_tasks, - beyonddb::CellStorage::new(replacement_client.clone(), "us-east-1"), - peer_directory.clone(), - vec!["123456789012".into()], - ) - .unwrap(); - let failover_id = loop { - let id = *uuid::Uuid::now_v7().as_bytes(); - let target = beyonddb::coordinator_target("123456789012", &id).unwrap(); - if CellAuthority::new(layout.clone()) - .load(target.cell_id()) - .await - .unwrap() - .is_none() - { - break id; - } - }; - let (failover_coordinator, _) = recovery::abandon_commit( - &remote_provisioner, - &replacement_client, - failover_id, - "serving-failover", - ) - .await; - tokio::time::sleep(std::time::Duration::from_secs(2)).await; - let live = CellAuthority::new(layout.clone()) - .load(failover_coordinator.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(live.value().owner.as_ref().unwrap().session, remote_session); + // Live-coordinator preservation and discovery after expiry are qualified + // separately by residency::live_owner_failover with an unfinished decision. public_server.abort(); remote_shutdown.send(()).unwrap(); remote_server.await.unwrap().unwrap(); remote_lease.cancel(); - tokio::time::timeout(std::time::Duration::from_secs(45), async { - loop { - if let Ok(status) = replacement_client - .query::( - &failover_coordinator, - None, - Json(beyonddb::ReadCrossCellTransactionInput { - account_id: "123456789012".into(), - transaction_id: failover_id, - routing_key: failover_id.to_vec(), - }), - ) - .await - { - let status = status.output.0.unwrap(); - assert_eq!(status.decision, beyonddb::CoordinatorDecision::Commit); - if status.resolved_count == 2 { - break; - } - } - tokio::time::sleep(std::time::Duration::from_millis(100)).await; - } - }) - .await - .expect("serving worker must discover and resolve the failed owner's transaction"); - for table in ["NetworkData", "RemoteTable"] { - let result = replacement_sdk - .get_item() - .table_name(table) - .key("id", AwsAttributeValue::S("serving-failover".into())) - .consistent_read(true) - .send() - .await - .unwrap(); - assert_eq!( - result.item().unwrap().get("value"), - Some(&AwsAttributeValue::S("recovered".into())) - ); - } // Recovery runs on the already-serving replacement. Empty journals must // not hide the failed index owner; its old image must survive takeover. index_recovery .assert_settled(&replacement_sdk, &replacement_client, "before") .await; index_recovery - .assert_owner(&CellAuthority::new(layout.clone()), replacement_session) + .assert_recovered_authority( + &CellAuthority::new(layout.clone()), + replacement_session, + &index_fences, + ) .await; peer_network::global_indexes::IndexRecovery::write(&replacement_sdk, "after").await; index_recovery diff --git a/tests/peer_network/global_indexes.rs b/tests/peer_network/global_indexes.rs index b279c00..ba714fc 100644 --- a/tests/peer_network/global_indexes.rs +++ b/tests/peer_network/global_indexes.rs @@ -2,14 +2,17 @@ use crate::*; use beyonddb::{ ReadPartitionIndexChange, RoutePageInput, RoutePageOutcome, data_target, global_index_target, }; -use cellule_runtime::identity::CellTarget; +use cellule_runtime::{ + control::{ControlState, OwnerFence}, + identity::CellTarget, +}; const ACCOUNT: &str = "123456789012"; const TABLE: &str = "ServingIndexFailover"; pub(crate) struct IndexRecovery { - targets: Vec, - table_id: String, + pub(crate) targets: Vec, + pub(crate) table_id: String, } impl IndexRecovery { @@ -151,10 +154,44 @@ impl IndexRecovery { .expect("signed index query and journal acknowledgement must converge"); } - pub(crate) async fn assert_owner(&self, authority: &CellAuthority, session: SessionId) { + pub(crate) async fn assert_owner( + &self, + authority: &CellAuthority, + session: SessionId, + ) -> Vec { + let mut fences = Vec::new(); for target in &self.targets { let current = authority.load(target.cell_id()).await.unwrap().unwrap(); assert_eq!(current.value().owner.as_ref().unwrap().session, session); + fences.push(current.value().owner_fence()); + } + fences + } + + pub(crate) async fn assert_recovered_authority( + &self, + authority: &CellAuthority, + session: SessionId, + before: &[OwnerFence], + ) { + assert_eq!(self.targets.len(), before.len()); + for (target, previous) in self.targets.iter().zip(before) { + let current = authority.load(target.cell_id()).await.unwrap().unwrap(); + let control = current.value(); + assert_eq!(control.incarnation, previous.incarnation); + assert!(control.epoch > previous.epoch, "owner epoch must advance"); + assert!( + control.root.is_some(), + "recovered root must remain published" + ); + assert!(control.recovery.is_none(), "recovery must finish"); + match control.state { + ControlState::Serving => { + assert_eq!(control.owner.as_ref().unwrap().session, session); + } + ControlState::Idle => assert!(control.owner.is_none()), + state => panic!("recovered Cell has invalid state: {state:?}"), + } } } } diff --git a/tests/peer_network/recovery.rs b/tests/peer_network/recovery.rs index 014d7c6..d2ac127 100644 --- a/tests/peer_network/recovery.rs +++ b/tests/peer_network/recovery.rs @@ -299,3 +299,34 @@ fn identity() -> cellule_runtime::MutationIdentity { expires_at_ms: issued_at_ms + 60_000, } } + +// Residency can change after recovery returns. Validate retained identity and +// durable progress while accepting either the live peer or this replacement. +pub(crate) fn assert_retained_range( + before: &cellule_runtime::control::Control, + after: &cellule_runtime::control::Control, + replacement: cellule_runtime::identity::SessionId, + live_peer: cellule_runtime::identity::SessionId, +) { + assert_eq!(after.cell, before.cell); + assert_eq!(after.incarnation, before.incarnation); + assert_eq!(after.code, before.code); + assert_eq!(after.schema, before.schema); + assert!(after.epoch >= before.epoch); + assert!( + after.root.as_ref().unwrap().commit_sequence + >= before.root.as_ref().unwrap().commit_sequence + ); + if let Some(owner) = &after.owner { + assert!(owner.session == replacement || owner.session == live_peer); + if before + .owner + .as_ref() + .is_none_or(|previous| previous.session != owner.session) + { + assert!(after.epoch > before.epoch); + } + } else { + assert_eq!(after.state, cellule_runtime::control::ControlState::Idle); + } +} diff --git a/tests/peer_network/residency.rs b/tests/peer_network/residency.rs index fdac658..a8a0558 100644 --- a/tests/peer_network/residency.rs +++ b/tests/peer_network/residency.rs @@ -1,18 +1,32 @@ +mod active_owners; +mod cell_models; mod codec; +mod coordinator_admission; +mod coordinator_discovery; +mod coordinator_registration; mod creation; +mod credential_failover; mod deletion; mod directories; mod discovery; +mod follower_durability; +mod forward_cache; mod index_splits; +mod live_owner_failover; mod ownership_race; mod placement; +mod prepare_batch; +mod prepare_capacity; +mod pressure; mod provisioning; mod rebalance; mod reclamation; mod recovery; +mod saved_images; mod splits; mod statistics; mod table_class; +mod update_batch; mod usage; use crate::*; @@ -61,6 +75,51 @@ impl Fixture { partitions: u16, store: Arc, cell_capacity: usize, + ) -> Self { + Self::with_store_capacity_and_peer_cache(partitions, store, cell_capacity, false).await + } + + async fn with_store_capacity_and_peer_cache( + partitions: u16, + store: Arc, + cell_capacity: usize, + peer_cache: bool, + ) -> Self { + Self::with_store_capacity_cache_and_router( + partitions, + store, + cell_capacity, + peer_cache, + std::convert::identity, + ) + .await + } + + async fn with_store_capacity_cache_and_router( + partitions: u16, + store: Arc, + cell_capacity: usize, + peer_cache: bool, + wrap: impl FnOnce(axum::Router) -> axum::Router, + ) -> Self { + Self::with_store_capacity_cache_router_and_telemetry( + partitions, + store, + cell_capacity, + peer_cache, + wrap, + None, + ) + .await + } + + async fn with_store_capacity_cache_router_and_telemetry( + partitions: u16, + store: Arc, + cell_capacity: usize, + peer_cache: bool, + wrap: impl FnOnce(axum::Router) -> axum::Router, + telemetry: Option>, ) -> Self { // SDK errors deliberately hide storage details. Retain server warnings // in the test output so CI failures identify the underlying boundary. @@ -114,6 +173,9 @@ impl Fixture { lease.clone(), ) .await; + if let Some(telemetry) = telemetry { + node.install_telemetry(telemetry).unwrap(); + } let peers = Arc::new( BeyonddbPeers::new(&node, layout.clone(), directory.clone(), session, &tls).unwrap(), ); @@ -156,7 +218,7 @@ impl Fixture { ) .await .unwrap(); - let client = peers.client(provisioner.clone()); + let client = peers.client_with_cache(provisioner.clone(), peer_cache); CellAuthorizationStore::new(client.clone()) .put_user_policy( "123456789012", @@ -174,7 +236,7 @@ impl Fixture { ) .await .unwrap(); - let router = peers.router(provisioner.clone()); + let router = wrap(peers.router_with_cache(provisioner.clone(), peer_cache)); let peer_server = tokio::spawn(async move { axum::serve( tls.listener(listener), diff --git a/tests/peer_network/residency/active_owners.rs b/tests/peer_network/residency/active_owners.rs new file mode 100644 index 0000000..5af2648 --- /dev/null +++ b/tests/peer_network/residency/active_owners.rs @@ -0,0 +1,254 @@ +use super::forward_cache::{CountedAuthority, CreationGate}; +use super::*; +use cellule_runtime::{ + cell::catalog::CatalogRole, + codec::{BoundedEncoder, WireValue}, + identity::{CellTarget, RequestId}, + peer::{PeerOperation, PeerRoundTrip, decode_peer_reply, wire}, +}; +use futures_util::TryStreamExt; +use object_store::ObjectStore; +use std::sync::atomic::Ordering; +use std::time::Duration; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn signed_forward_cache_keeps_hydrating_owners_without_authority_reads() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let original = &fixture.data[0].0; + let target = CellTarget::new( + account_target("123456789012").unwrap().tenant(), + beyonddb::APPLICATION_ID, + original.catalog().entry().namespace(), + original.catalog().entry().partition(), + ) + .unwrap(); + // A real published 8 MiB database leaves background hydration to do after + // cold activation. The SDK item and its normal partition schema remain. + let now = now_ms(); + original + .execute( + cellule_runtime::MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([202; 32]), + now, + 1_024, + 64, + |transaction| { + transaction.execute_batch( + "CREATE TABLE hydration_payload(value BLOB NOT NULL); \ + WITH RECURSIVE numbers(value) AS ( \ + SELECT 1 UNION ALL SELECT value + 1 FROM numbers WHERE value < 512 \ + ) \ + INSERT INTO hydration_payload(value) SELECT zeroblob(16384) FROM numbers;", + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + vec![], + )) + }, + ) + .await + .unwrap(); + original.drain().await.unwrap(); + let remote = provisioning::Remote::with_cache(&fixture).await; + let table = fixture + .client + .query::( + &account_target("123456789012").unwrap(), + None, + Json("Residency".into()), + ) + .await + .unwrap() + .output + .0 + .unwrap(); + let route = crate::single_leaf_route( + &fixture.client, + &account_target("123456789012").unwrap(), + &table.id, + ) + .await + .unwrap(); + let restored = remote + .provisioner + .admit_existing_partition("123456789012", &table.id, &route.partitions[0].partition_id) + .await + .unwrap(); + assert_eq!(restored.cell_id(), target.cell_id()); + let body_paths = store + .list(None) + .try_collect::>() + .await + .unwrap() + .into_iter() + .filter(|object| { + object.location.as_ref().ends_with(".ltx") + || object.location.as_ref().ends_with(".bundle") + }) + .map(|object| object.location) + .collect(); + let gate = CreationGate { + paths: body_paths, + entered: Arc::new(tokio::sync::Semaphore::new(0)), + release: Arc::new(tokio::sync::Semaphore::new(0)), + }; + *store.read_gate.lock().unwrap() = Some(gate.clone()); + let reached = tokio::time::timeout(Duration::from_secs(3), gate.entered.acquire()).await; + let observation = if let Ok(permit) = reached { + permit.unwrap().forget(); + let runtime = remote.node.runtime(); + let active = runtime + .active_handle(&target, CatalogRole::Sql) + .await + .unwrap(); + let resident = runtime + .resident_handle(&target, CatalogRole::Sql) + .await + .unwrap(); + let jobs = runtime.stats().hydration_jobs(); + // Sign with the ingress identity while targeting the exact known remote + // node. This isolates receiver authority reads from sender discovery. + let tls = LoadedPeerTls::load( + &fixture._files.path().join("owner.crt"), + &fixture._files.path().join("owner.key"), + &fixture._files.path().join("ca.crt"), + "localhost", + ) + .unwrap(); + let peers = PeerHttpRoundTrip::new( + Arc::new(BeyonddbPeerScope), + CellAuthority::new(fixture.layout.clone()), + fixture.directory.clone(), + Arc::new(tls.client_identity()), + fixture.session, + ); + let signer = PeerSigner::new( + fixture.session, + fixture.application.registry().release_digest(), + tls.signing_key().clone(), + ); + let principal = PeerPrincipal { + issuer: format!( + "beyonddb-peer:{}", + fixture + .directory + .fleet() + .as_bytes() + .iter() + .map(|b| format!("{b:02x}")) + .collect::() + ), + subject: fixture + .session + .as_bytes() + .iter() + .map(|b| format!("{b:02x}")) + .collect(), + actions: vec!["beyonddb.cell.invoke".into()], + }; + let destination = fixture + .directory + .load(remote.session, now_ms()) + .await + .unwrap() + .unwrap() + .advertisement() + .clone(); + *store.path.lock().unwrap() = + Some(fixture.layout.control_path(target.cell_id().as_bytes())); + store.reads.store(0, Ordering::SeqCst); + let mut encoder = BoundedEncoder::new(64).unwrap(); + Json(()).encode(&mut encoder).unwrap(); + let input = encoder.finish(); + let mut outcomes = Vec::new(); + for request_index in 0..5 { + if request_index == 4 { + tokio::time::sleep(Duration::from_millis(550)).await; + } + let now = now_ms(); + let request = signer + .sign( + principal.clone(), + now, + now + 60_000, + 30_000, + PeerOperation::Read(wire::ReadRequest { + target: Some(wire::Target { + tenant_id: target.tenant().as_bytes().to_vec(), + application_id: target.application().as_bytes().to_vec(), + namespace_id: target.namespace().as_bytes().to_vec(), + partition: target.partition().to_vec(), + }), + timeout_ms: 30_000, + minimum: None, + expected: Some(wire::CellDescription { + cell_id: restored.cell_id().as_bytes().to_vec(), + incarnation: restored.incarnation().as_bytes().to_vec(), + code: restored.code().as_bytes().to_vec(), + schema: restored.schema(), + }), + operation: Some(wire::read_request::Operation::CellQuery( + wire::CellQuery { + query_id: 4, + codec_version: 1, + input: input.clone(), + }, + )), + }), + ) + .unwrap(); + outcomes.push( + tokio::time::timeout( + Duration::from_secs(1), + peers.send_to_node(target.clone(), destination.clone(), request, 30_000), + ) + .await, + ); + } + Some(( + active.map(|handle| handle.owner_fence()), + resident.is_some(), + jobs, + outcomes, + store.reads.load(Ordering::SeqCst), + )) + } else { + None + }; + *store.read_gate.lock().unwrap() = None; + gate.release.add_permits(1); + *store.path.lock().unwrap() = None; + let expected = fixture.data[0].1.clone(); + let item = provisioning::sdk_without_retries(&fixture) + .get_item() + .table_name("Residency") + .key("id", expected["id"].clone()) + .consistent_read(true) + .send() + .await; + let fence = restored.owner_fence(); + remote.shutdown().await; + fixture.shutdown().await; + let (active, resident, jobs, outcomes, reads) = + observation.expect("cold owner did not reach held background hydration"); + assert_eq!(active, Some(fence)); + assert!(!resident); + assert_eq!(jobs, 1); + for outcome in outcomes { + let reply = outcome + .expect("forwarded read waited for hydration") + .unwrap(); + let decoded = decode_peer_reply(&reply).unwrap(); + assert!( + matches!(decoded.outcome, Some(wire::peer_reply::Outcome::Read(_))), + "forwarded outcome: {decoded:?}" + ); + } + assert_eq!(item.unwrap().item, Some(expected)); + println!("five signed hydrating-owner reads including expired cache: authority reads={reads}"); + assert_eq!(reads, 0, "hydrating owner fell back to metadata lookup"); +} diff --git a/tests/peer_network/residency/cell_models.rs b/tests/peer_network/residency/cell_models.rs new file mode 100644 index 0000000..030c0fa --- /dev/null +++ b/tests/peer_network/residency/cell_models.rs @@ -0,0 +1,1111 @@ +use super::provisioning::{create, sdk_without_retries, table_id}; +use super::*; +use aws_sdk_dynamodb::types::{Get, Put, Tag, TransactGetItem, TransactWriteItem}; +use beyonddb::{RoutePageInput, RoutePageOutcome}; +use extenddb_storage::StreamEngine; + +const ACCOUNT: &str = "123456789012"; + +#[derive(Default)] +struct DataCommands(std::sync::atomic::AtomicU64); + +impl cellule_runtime::fleet::telemetry::CellTelemetry for DataCommands { + fn primitive_operation( + &self, + module: &'static str, + kind: cellule_runtime::fleet::telemetry::PrimitiveOperationKind, + _: cellule_runtime::fleet::telemetry::PrimitiveOperationOutcome, + _: std::time::Duration, + ) { + if module == "beyonddb-data" + && kind == cellule_runtime::fleet::telemetry::PrimitiveOperationKind::Command + { + self.0.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sdk_returned_updates_coalesce_and_survive_owner_restore() { + let commands = Arc::new(DataCommands::default()); + let fixture = Fixture::with_store_capacity_cache_router_and_telemetry( + 1, + Arc::new(InMemory::new()), + 16, + false, + std::convert::identity, + Some(commands.clone()), + ) + .await; + let remote = super::provisioning::Remote::new(&fixture).await; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("http://{}", listener.local_addr().unwrap()); + let state = build_http_state( + &fixture.node, + remote.client(&fixture), + fixture.layout.clone(), + fixture.provisioner.clone(), + [38; 32], + "us-east-1", + endpoint.clone(), + ) + .unwrap(); + let server = tokio::spawn(async move { + extenddb_server::start_server(listener, state, None, None) + .await + .unwrap(); + }); + let sdk = aws_sdk_dynamodb::Client::from_conf( + sdk_without_retries(&fixture) + .config() + .to_builder() + .endpoint_url(endpoint) + .build(), + ); + // Warm the directory/auth paths before measuring data commands. Every + // update below uses a distinct key and requests its committed new image. + sdk.get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("returned-warmup".into())) + .send() + .await + .unwrap(); + commands.0.store(0, std::sync::atomic::Ordering::Relaxed); + let mut requests = tokio::task::JoinSet::new(); + let start = Arc::new(tokio::sync::Barrier::new(65)); + for index in 0..64 { + let sdk = sdk.clone(); + let start = start.clone(); + requests.spawn(async move { + start.wait().await; + let item = sdk + .update_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S(format!("returned-{index}"))) + .update_expression("ADD #value :one") + .expression_attribute_names("#value", "value") + .expression_attribute_values(":one", AwsAttributeValue::N("1".into())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllNew) + .send() + .await + .unwrap(); + let attributes = item.attributes.unwrap(); + assert_eq!(attributes["value"], AwsAttributeValue::N("1".into())); + assert_eq!( + attributes["id"], + AwsAttributeValue::S(format!("returned-{index}")) + ); + }); + } + start.wait().await; + while let Some(result) = requests.join_next().await { + result.unwrap(); + } + let count = commands.0.load(std::sync::atomic::Ordering::Relaxed); + eprintln!("64 returned SDK updates used {count} durable data commands"); + assert!(count < 64, "returned updates should share durable commands"); + // Concurrent requests for one key must be deferred into separate commands + // and retain each ADD's own committed result. + let mut requests = tokio::task::JoinSet::new(); + for _ in 0..8 { + let sdk = sdk.clone(); + requests.spawn(async move { + sdk.update_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("returned-repeated".into())) + .update_expression("ADD #value :one") + .expression_attribute_names("#value", "value") + .expression_attribute_values(":one", AwsAttributeValue::N("1".into())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllNew) + .send() + .await + .unwrap() + .attributes + .unwrap()["value"] + .as_n() + .unwrap() + .parse::() + .unwrap() + }); + } + let mut values = std::collections::BTreeSet::new(); + while let Some(result) = requests.join_next().await { + values.insert(result.unwrap()); + } + assert_eq!(values, (1..=8).collect()); + let id = table_id(&fixture, "Residency").await; + let range = ranges(&fixture, "Residency").await.remove(0); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + for index in 0..64 { + let item = sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S(format!("returned-{index}"))) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::N("1".into())); + } + server.abort(); + let _ = server.await; + remote.shutdown().await; + fixture.shutdown().await; +} + +#[derive(Default)] +struct CoordinatorQueries(std::sync::atomic::AtomicU64); + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sdk_transaction_prepares_coalesce_and_survive_owner_restore() { + transaction_prepares(false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sdk_transaction_prepares_recover_lost_batch_reply() { + transaction_prepares(true).await; +} + +async fn transaction_prepares(drop_reply: bool) { + let commands = Arc::new(DataCommands::default()); + let dropped = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let drop_marker = dropped.clone(); + let fixture = Fixture::with_store_capacity_cache_router_and_telemetry( + 1, + Arc::new(InMemory::new()), + 128, + false, + move |router| { + router.layer(axum::middleware::from_fn( + move |request: axum::extract::Request, next: axum::middleware::Next| { + let dropped = drop_marker.clone(); + async move { + use cellule_runtime::peer::wire::{self, mutation_request, peer_request}; + use prost::Message; + let (parts, body) = request.into_parts(); + let bytes = axum::body::to_bytes(body, 8 * 1024 * 1024).await.unwrap(); + let batch = wire::PeerRequest::decode(bytes.as_ref()).ok().is_some_and(|request| { + matches!(request.operation, + Some(peer_request::Operation::Mutate(mutation)) + if matches!(&mutation.operation, + Some(mutation_request::Operation::CellCommand(command)) if command.command_id == 29)) + }); + let request = axum::extract::Request::from_parts(parts, axum::body::Body::from(bytes)); + let response = next.run(request).await; + if drop_reply && batch && !dropped.swap(true, std::sync::atomic::Ordering::SeqCst) { + // The real authenticated handler has durably published. + // Lose only the reply body, after dispatch, so the + // caller must observe each intent instead of retrying. + return axum::response::Response::new(axum::body::Body::from_stream( + futures_util::stream::once(async { + Err::(std::io::Error::from(std::io::ErrorKind::ConnectionReset)) + }), + )); + } + response + } + }, + )) + }, + Some(commands.clone()), + ) + .await; + let remote = super::provisioning::Remote::new(&fixture).await; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("http://{}", listener.local_addr().unwrap()); + let state = build_http_state( + &fixture.node, + remote.client(&fixture), + fixture.layout.clone(), + fixture.provisioner.clone(), + [38; 32], + "us-east-1", + endpoint.clone(), + ) + .unwrap(); + let server = tokio::spawn(async move { + extenddb_server::start_server(listener, state, None, None) + .await + .unwrap(); + }); + let sdk = aws_sdk_dynamodb::Client::from_conf( + sdk_without_retries(&fixture) + .config() + .to_builder() + .endpoint_url(endpoint) + .build(), + ); + // Pre-admit the exact token shards so discovery does not stagger the + // concurrent prepares. Requests still traverse signed SDK/HTTP/TLS paths. + for index in 0..32 { + beyonddb::CoordinatorProvisioner::ensure( + fixture.provisioner.as_ref(), + &fixture.client, + ACCOUNT, + format!("prepare-batch-{index}").as_bytes(), + ) + .await + .unwrap(); + } + sdk.get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("prepare-warmup".into())) + .send() + .await + .unwrap(); + commands.0.store(0, std::sync::atomic::Ordering::Relaxed); + let start = Arc::new(tokio::sync::Barrier::new(33)); + let mut requests = tokio::task::JoinSet::new(); + for index in 0..32 { + let sdk = sdk.clone(); + let start = start.clone(); + requests.spawn(async move { + let request = sdk + .transact_write_items() + .client_request_token(format!("prepare-batch-{index}")) + .set_transact_items(Some( + (0..2) + .map(|item| { + TransactWriteItem::builder() + .put( + Put::builder() + .table_name("Residency") + .item( + "id", + AwsAttributeValue::S(format!("prepare-{index}-{item}")), + ) + .item("value", AwsAttributeValue::N(index.to_string())) + .build() + .unwrap(), + ) + .build() + }) + .collect(), + )); + start.wait().await; + request.send().await.unwrap(); + }); + } + start.wait().await; + while let Some(result) = requests.join_next().await { + result.unwrap(); + } + let count = commands.0.load(std::sync::atomic::Ordering::Relaxed); + eprintln!("32 signed transaction writes used {count} durable data commands"); + let id = table_id(&fixture, "Residency").await; + let range = ranges(&fixture, "Residency").await.remove(0); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + for index in 0..32 { + for item in 0..2 { + let restored = sdk + .get_item() + .table_name("Residency") + .key( + "id", + AwsAttributeValue::S(format!("prepare-{index}-{item}")), + ) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(restored["value"], AwsAttributeValue::N(index.to_string())); + } + } + server.abort(); + let _ = server.await; + remote.shutdown().await; + fixture.shutdown().await; + assert_eq!( + dropped.load(std::sync::atomic::Ordering::SeqCst), + drop_reply + ); + assert!( + count < 64, + "independent prepares must share durable commits" + ); +} + +impl cellule_runtime::fleet::telemetry::CellTelemetry for CoordinatorQueries { + fn primitive_operation( + &self, + module: &'static str, + kind: cellule_runtime::fleet::telemetry::PrimitiveOperationKind, + _: cellule_runtime::fleet::telemetry::PrimitiveOperationOutcome, + _: std::time::Duration, + ) { + if module == "beyonddb-coordinator" + && kind == cellule_runtime::fleet::telemetry::PrimitiveOperationKind::Query + { + self.0.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_fresh_read_reuses_acknowledged_participants() { + let queries = Arc::new(CoordinatorQueries::default()); + let fixture = Fixture::with_store_capacity_cache_router_and_telemetry( + 2, + Arc::new(InMemory::new()), + 16, + false, + std::convert::identity, + Some(queries.clone()), + ) + .await; + let remote = super::provisioning::Remote::new(&fixture).await; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("http://{}", listener.local_addr().unwrap()); + let state = build_http_state( + &fixture.node, + remote.client(&fixture), + fixture.layout.clone(), + fixture.provisioner.clone(), + [38; 32], + "us-east-1", + endpoint.clone(), + ) + .unwrap(); + let server = tokio::spawn(async move { + extenddb_server::start_server(listener, state, None, None) + .await + .unwrap(); + }); + let sdk = aws_sdk_dynamodb::Client::from_conf( + sdk_without_retries(&fixture) + .config() + .to_builder() + .endpoint_url(endpoint) + .build(), + ); + assert_eq!(fixture.data.len(), 2); + // Reverse request order relative to the durable participant ordering. The + // third key is absent: a saved None image must still fill its result slot. + let mut keys = fixture + .data + .iter() + .rev() + .map(|(_, item)| item["id"].clone()) + .collect::>(); + keys.push(AwsAttributeValue::S("fresh-read-missing".into())); + let read = sdk.transact_get_items().set_transact_items(Some( + keys.iter() + .map(|key| { + TransactGetItem::builder() + .get( + Get::builder() + .table_name("Residency") + .key("id", key.clone()) + .build() + .unwrap(), + ) + .build() + }) + .collect(), + )); + queries.0.store(0, std::sync::atomic::Ordering::Relaxed); + let result = read.clone().send().await.unwrap(); + let query_count = queries.0.load(std::sync::atomic::Ordering::Relaxed); + eprintln!("fresh signed two-Cell read used {query_count} coordinator queries"); + let images = result.responses.unwrap(); + assert_eq!(images.len(), 3); + for (image, (_, expected)) in images.iter().zip(fixture.data.iter().rev()) { + assert_eq!(image.item.as_ref(), Some(expected)); + } + assert!(images[2].item.is_none()); + // Completed read cleanup must leave no locks after owner restoration. + for (handle, _) in &fixture.data { + handle.drain().await.unwrap(); + } + let id = table_id(&fixture, "Residency").await; + for range in ranges(&fixture, "Residency").await { + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + } + for (index, key) in keys.iter().take(2).enumerate() { + sdk.put_item() + .table_name("Residency") + .item("id", key.clone()) + .item("value", AwsAttributeValue::N(index.to_string())) + .send() + .await + .unwrap(); + } + let restored = read.send().await.unwrap().responses.unwrap(); + for (index, image) in restored.iter().take(2).enumerate() { + assert_eq!( + image.item.as_ref().unwrap()["value"], + AwsAttributeValue::N(index.to_string()) + ); + } + assert!(restored[2].item.is_none()); + // Large BEGIN payloads retain durable participant discovery. These legal + // partition keys exceed the shortcut's 32 KiB bound in aggregate. + queries.0.store(0, std::sync::atomic::Ordering::Relaxed); + let large = sdk + .transact_get_items() + .set_transact_items(Some( + (0..40) + .map(|index| { + TransactGetItem::builder() + .get( + Get::builder() + .table_name("Residency") + .key( + "id", + AwsAttributeValue::S(format!( + "wide-read-{index}-{}", + "x".repeat(1_000) + )), + ) + .build() + .unwrap(), + ) + .build() + }) + .collect(), + )) + .send() + .await + .unwrap() + .responses + .unwrap(); + assert_eq!(large.len(), 40); + assert!(large.iter().all(|image| image.item.is_none())); + assert!( + queries.0.load(std::sync::atomic::Ordering::Relaxed) > 0, + "oversized BEGIN must retain durable discovery" + ); + server.abort(); + let _ = server.await; + remote.shutdown().await; + fixture.shutdown().await; + assert_eq!( + query_count, 3, + "acknowledged BEGIN removes participant rediscovery; canonical completion retains three status/discovery queries" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_fresh_transaction_reuses_acknowledged_write_completion() { + let queries = Arc::new(CoordinatorQueries::default()); + let fixture = Fixture::with_store_capacity_cache_router_and_telemetry( + 2, + Arc::new(InMemory::new()), + 16, + false, + std::convert::identity, + Some(queries.clone()), + ) + .await; + let remote = super::provisioning::Remote::new(&fixture).await; + beyonddb::CoordinatorProvisioner::ensure( + fixture.provisioner.as_ref(), + &fixture.client, + ACCOUNT, + b"fresh-begin-payload", + ) + .await + .unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("http://{}", listener.local_addr().unwrap()); + let state = build_http_state( + &fixture.node, + remote.client(&fixture), + fixture.layout.clone(), + fixture.provisioner.clone(), + [38; 32], + "us-east-1", + endpoint.clone(), + ) + .unwrap(); + let server = tokio::spawn(async move { + extenddb_server::start_server(listener, state, None, None) + .await + .unwrap(); + }); + let sdk = aws_sdk_dynamodb::Client::from_conf( + sdk_without_retries(&fixture) + .config() + .to_builder() + .endpoint_url(endpoint) + .build(), + ); + queries.0.store(0, std::sync::atomic::Ordering::Relaxed); + let keys = fixture + .data + .iter() + .map(|(_, item)| item["id"].clone()) + .collect::>(); + assert_eq!( + keys.len(), + 2, + "the write must reach two distinct data Cells" + ); + let write = sdk + .transact_write_items() + .client_request_token("fresh-begin-payload") + .set_transact_items(Some( + keys.iter() + .map(|key| { + TransactWriteItem::builder() + .put( + Put::builder() + .table_name("Residency") + .item("id", key.clone()) + .item("value", AwsAttributeValue::N("7".into())) + .build() + .unwrap(), + ) + .build() + }) + .collect(), + )); + write.clone().send().await.unwrap(); + assert_eq!( + queries.0.load(std::sync::atomic::Ordering::Relaxed), + 1, + "fresh write needs token lookup; acknowledged BEGIN, decision and resolution receipts prove completion" + ); + let coordinator = beyonddb::coordinator_target(ACCOUNT, b"fresh-begin-payload").unwrap(); + let pending = fixture + .client + .query::( + &coordinator, + None, + beyonddb::Json(beyonddb::ReadPendingCrossCellTransactionsInput { + after: None, + limit: 100, + }), + ) + .await + .unwrap() + .output + .0; + assert!( + pending.is_empty(), + "successful SDK completion must durably record every resolution" + ); + for (handle, _) in &fixture.data { + handle.drain().await.unwrap(); + } + fixture + .provisioner + .admit_coordinator(ACCOUNT, b"fresh-begin-payload") + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_coordinator(ACCOUNT, b"fresh-begin-payload") + .await + .unwrap(); + let id = table_id(&fixture, "Residency").await; + for range in ranges(&fixture, "Residency").await { + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + } + for key in &keys { + let item = sdk + .get_item() + .table_name("Residency") + .key("id", key.clone()) + .consistent_read(true) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::N("7".into())); + } + sdk.put_item() + .table_name("Residency") + .item("id", keys[0].clone()) + .item("value", AwsAttributeValue::N("8".into())) + .send() + .await + .unwrap(); + queries.0.store(0, std::sync::atomic::Ordering::Relaxed); + write.send().await.unwrap(); + assert_eq!( + queries.0.load(std::sync::atomic::Ordering::Relaxed), + 3, + "token replay must read durable coordinator state" + ); + let item = sdk + .get_item() + .table_name("Residency") + .key("id", keys[0].clone()) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::N("8".into())); + beyonddb::CoordinatorProvisioner::ensure( + fixture.provisioner.as_ref(), + &fixture.client, + ACCOUNT, + b"large-begin-payload", + ) + .await + .unwrap(); + queries.0.store(0, std::sync::atomic::Ordering::Relaxed); + let padding = "\0".repeat(6_000); + sdk.transact_write_items() + .client_request_token("large-begin-payload") + .transact_items( + TransactWriteItem::builder() + .put( + Put::builder() + .table_name("Residency") + .item("id", AwsAttributeValue::S("large-begin".into())) + .item("padding", AwsAttributeValue::S(padding.clone())) + .build() + .unwrap(), + ) + .build(), + ) + .send() + .await + .unwrap(); + assert_eq!( + queries.0.load(std::sync::atomic::Ordering::Relaxed), + 7, + "escaped inputs beyond the reuse bound retain durable chunk discovery" + ); + let large_item = sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("large-begin".into())) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(large_item["padding"], AwsAttributeValue::S(padding)); + server.abort(); + let _ = server.await; + remote.shutdown().await; + fixture.shutdown().await; +} + +fn model_tag(value: &str) -> Tag { + Tag::builder() + .key("beyonddb:cell-model") + .value(value) + .build() + .unwrap() +} + +pub(super) async fn ranges(fixture: &Fixture, table: &str) -> Vec { + let id = table_id(fixture, table).await; + let page = beyonddb::read_route_page( + &fixture.client, + &account_target(ACCOUNT).unwrap(), + RoutePageInput { + table_id: id, + start_hash: None, + after_lower: None, + expected_epoch: None, + }, + ) + .await + .unwrap(); + let RoutePageOutcome::Page { + partitions, + has_more, + .. + } = page + else { + panic!("table route is missing"); + }; + assert!(!has_more); + partitions +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_cell_models_select_persisted_table_placement() { + let fixture = Fixture::with_capacity(4, 32).await; + let sdk = sdk_without_retries(&fixture); + for (name, model, expected) in [ + ("ResidencyModelSingle", "single", 1), + ("ResidencyModelAuto", "auto", 1), + ("ResidencyModelPartitioned", "partitioned", 4), + ] { + create(&sdk, name, false) + .tags(model_tag(model)) + .send() + .await + .unwrap(); + assert_eq!(ranges(&fixture, name).await.len(), expected, "{model}"); + } + create(&sdk, "ResidencyModelDefault", false) + .send() + .await + .unwrap(); + assert_eq!(ranges(&fixture, "ResidencyModelDefault").await.len(), 4); + let error = create(&sdk, "ResidencyModelInvalid", false) + .tags(model_tag("unknown")) + .send() + .await + .unwrap_err(); + assert_eq!( + error.as_service_error().unwrap().code(), + Some("ValidationException") + ); + assert!( + sdk.describe_table() + .table_name("ResidencyModelInvalid") + .send() + .await + .is_err() + ); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_single_cell_resists_splits_and_restores_atomic_reads() { + let fixture = Fixture::with_capacity(2, 24).await; + let sdk = sdk_without_retries(&fixture); + let stream_arn = create(&sdk, "ResidencyModelSingle", false) + .tags(model_tag("single")) + .stream_specification( + aws_sdk_dynamodb::types::StreamSpecification::builder() + .stream_enabled(true) + .stream_view_type(aws_sdk_dynamodb::types::StreamViewType::NewAndOldImages) + .build() + .unwrap(), + ) + .send() + .await + .unwrap() + .table_description + .unwrap() + .latest_stream_arn + .unwrap(); + let id = table_id(&fixture, "ResidencyModelSingle").await; + for key in ["first", "second"] { + sdk.put_item() + .table_name("ResidencyModelSingle") + .item("id", AwsAttributeValue::S(key.into())) + .item("value", AwsAttributeValue::N("7".into())) + .send() + .await + .unwrap(); + } + let write = sdk + .transact_write_items() + .client_request_token("single-cell-restart-token") + .transact_items( + TransactWriteItem::builder() + .put( + Put::builder() + .table_name("ResidencyModelSingle") + .item("id", AwsAttributeValue::S("first".into())) + .item("value", AwsAttributeValue::N("8".into())) + .build() + .unwrap(), + ) + .build(), + ); + write.clone().send().await.unwrap(); + let range = ranges(&fixture, "ResidencyModelSingle").await.remove(0); + assert!( + fixture + .provisioner + .split_if_over_database_bytes( + ACCOUNT, + fixture.client.clone(), + &id, + range.partition_id, + range.lower, + 1, + ) + .await + .unwrap() + .is_none() + ); + assert!( + fixture + .provisioner + .split_partition( + ACCOUNT, + fixture.client.clone(), + &id, + range.partition_id, + range.lower, + ) + .await + .is_err() + ); + let target = beyonddb::data_target(ACCOUNT, &id, &range.partition_id).unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + sdk.put_item() + .table_name("ResidencyModelSingle") + .item("id", AwsAttributeValue::S("first".into())) + .item("value", AwsAttributeValue::N("9".into())) + .send() + .await + .unwrap(); + // The replay must not overwrite a newer item after participant restoration. + write.send().await.unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let before = authority.load(target.cell_id()).await.unwrap().unwrap(); + let request = sdk.transact_get_items().set_transact_items(Some( + ["second", "first"] + .into_iter() + .map(|key| { + TransactGetItem::builder() + .get( + Get::builder() + .table_name("ResidencyModelSingle") + .key("id", AwsAttributeValue::S(key.into())) + .build() + .unwrap(), + ) + .build() + }) + .collect(), + )); + let result = request.send().await.unwrap(); + assert_eq!(result.responses().len(), 2); + assert_eq!( + result.responses()[0].item().unwrap()["id"], + AwsAttributeValue::S("second".into()) + ); + assert_eq!( + result.responses()[1].item().unwrap()["id"], + AwsAttributeValue::S("first".into()) + ); + assert_eq!( + result.responses()[1].item().unwrap()["value"], + AwsAttributeValue::N("9".into()) + ); + let after = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!( + before.value().root.as_ref().unwrap().commit_sequence, + after.value().root.as_ref().unwrap().commit_sequence, + "a small single-Cell transaction read must not commit participant work" + ); + assert_eq!(ranges(&fixture, "ResidencyModelSingle").await.len(), 1); + let streams = beyonddb::CellStorage::new(fixture.client.clone(), "us-east-1"); + let description = streams + .describe_stream( + ACCOUNT, + &extenddb_core::types::DescribeStreamInput { + stream_arn: stream_arn.clone(), + limit: Some(100), + exclusive_start_shard_id: None, + }, + ) + .await + .unwrap(); + assert_eq!(description.shards.len(), 1); + let shard = &description.shards[0].shard_id; + streams + .validate_shard(ACCOUNT, &stream_arn, shard) + .await + .unwrap(); + let (records, _) = streams + .get_stream_records(ACCOUNT, shard, None, 100) + .await + .unwrap(); + assert_eq!( + records.len(), + 4, + "token replay must not emit another record" + ); + // Auto starts at one Cell but remains eligible for the same durable split path. + create(&sdk, "ResidencyModelAuto", false) + .tags(model_tag("auto")) + .send() + .await + .unwrap(); + let id = table_id(&fixture, "ResidencyModelAuto").await; + let range = ranges(&fixture, "ResidencyModelAuto").await.remove(0); + assert!( + fixture + .provisioner + .split_if_over_database_bytes( + ACCOUNT, + fixture.client.clone(), + &id, + range.partition_id, + range.lower, + 1, + ) + .await + .unwrap() + .is_some() + ); + assert_eq!(ranges(&fixture, "ResidencyModelAuto").await.len(), 2); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_single_cell_keeps_index_growth_and_ignores_later_placement_tags() { + let fixture = Fixture::with_capacity(4, 32).await; + let sdk = sdk_without_retries(&fixture); + const NAME: &str = "ResidencyModelIndex"; + let created = create(&sdk, NAME, true) + .tags(model_tag("single")) + .send() + .await + .unwrap() + .table_description + .unwrap(); + // Pinned ExtendDB authorizes TagResource against table/* rather than the + // supplied ARN. Keep that existing protocol limitation explicit here. + CellAuthorizationStore::new(fixture.client.clone()) + .put_user_policy( + ACCOUNT, + "network-user", + "model-tags", + &serde_json::json!({"Version":"2012-10-17", "Statement":[{ + "Effect":"Allow", "Action":"dynamodb:TagResource", "Resource":"*" + }]}) + .to_string(), + ) + .await + .unwrap(); + sdk.tag_resource() + .resource_arn(created.table_arn().unwrap()) + .tags(model_tag("partitioned")) + .send() + .await + .unwrap(); + let table = fixture + .client + .query::(&account_target(ACCOUNT).unwrap(), None, Json(NAME.into())) + .await + .unwrap() + .output + .0 + .unwrap(); + assert_eq!(table.placement, beyonddb::TablePlacement::Single); + assert_eq!(ranges(&fixture, NAME).await.len(), 1); + let index = &table.global_secondary_indexes[0]; + let index_ranges = beyonddb::read_route_page( + &fixture.client, + &account_target(ACCOUNT).unwrap(), + RoutePageInput { + table_id: index.id.clone(), + start_hash: None, + after_lower: None, + expected_epoch: None, + }, + ) + .await + .unwrap(); + let RoutePageOutcome::Page { + partitions, + has_more, + .. + } = index_ranges + else { + panic!("index route missing"); + }; + assert!(!has_more); + assert_eq!(partitions.len(), 1); + beyonddb::CellStorage::new(fixture.client.clone(), "us-east-1") + .install_global_index_loop( + &fixture.tasks, + vec![ACCOUNT.into()], + fixture.provisioner.clone(), + fixture.directory.clone(), + ) + .unwrap(); + sdk.put_item() + .table_name(NAME) + .item("id", AwsAttributeValue::S("indexed".into())) + .item("bucket", AwsAttributeValue::S("bucket".into())) + .send() + .await + .unwrap(); + let query = sdk + .query() + .table_name(NAME) + .index_name("ByBucket") + .key_condition_expression("#bucket = :bucket") + .expression_attribute_names("#bucket", "bucket") + .expression_attribute_values(":bucket", AwsAttributeValue::S("bucket".into())); + tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + if query.clone().send().await.unwrap().items().len() == 1 { + break; + } + tokio::time::sleep(std::time::Duration::from_millis(25)).await; + } + }) + .await + .unwrap(); + let range = &partitions[0]; + assert!( + fixture + .provisioner + .split_global_index_if_over_database_bytes( + ACCOUNT, + fixture.client.clone(), + &index.id, + range.partition_id, + range.lower, + 1 + ) + .await + .unwrap() + .is_some() + ); + assert_eq!( + query.send().await.unwrap().items()[0]["id"], + AwsAttributeValue::S("indexed".into()) + ); + assert_eq!(ranges(&fixture, NAME).await.len(), 1); + fixture.shutdown().await; +} diff --git a/tests/peer_network/residency/coordinator_admission.rs b/tests/peer_network/residency/coordinator_admission.rs new file mode 100644 index 0000000..03e3d37 --- /dev/null +++ b/tests/peer_network/residency/coordinator_admission.rs @@ -0,0 +1,700 @@ +use super::forward_cache::CountedAuthority; +use super::*; +use beyonddb::{ + CoordinatorProvisioner, ReadCoordinatorRegistration, RegisterCoordinatorShardInput, +}; +use cellule_runtime::cell::catalog::CellCatalog; +use std::sync::atomic::Ordering; + +const ACCOUNT: &str = "123456789012"; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn concurrent_sdk_token_replay_serializes_one_cold_coordinator_and_survives_restore() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let token = "same-cold-coordinator"; + let target = beyonddb::coordinator_target(ACCOUNT, token.as_bytes()).unwrap(); + let entered = Arc::new(tokio::sync::Semaphore::new(0)); + let release = Arc::new(tokio::sync::Semaphore::new(0)); + *store.creation_gate.lock().unwrap() = Some(super::forward_cache::CreationGate { + paths: [fixture.layout.control_path(target.cell_id().as_bytes())] + .into_iter() + .collect(), + entered: entered.clone(), + release: release.clone(), + }); + let request = fixture + .sdk + .transact_write_items() + .client_request_token(token) + .transact_items( + aws_sdk_dynamodb::types::TransactWriteItem::builder() + .put( + aws_sdk_dynamodb::types::Put::builder() + .table_name("Residency") + .item("id", AwsAttributeValue::S("same-cold".into())) + .item("value", AwsAttributeValue::N("7".into())) + .build() + .unwrap(), + ) + .build(), + ); + let first = tokio::spawn(request.clone().send()); + tokio::time::timeout(std::time::Duration::from_secs(2), entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let second = tokio::spawn(request.clone().send()); + let serialized = tokio::time::timeout(std::time::Duration::from_millis(150), async { + entered.acquire().await.unwrap().forget(); + }) + .await + .is_err(); + release.add_permits(2); + for job in [first, second] { + tokio::time::timeout(std::time::Duration::from_secs(10), job) + .await + .unwrap() + .unwrap() + .unwrap(); + } + assert!( + serialized, + "the same coordinator was initialized concurrently" + ); + let before = CellAuthority::new(fixture.layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .incarnation; + fixture + .provisioner + .admit_coordinator(ACCOUNT, token.as_bytes()) + .await + .unwrap() + .drain() + .await + .unwrap(); + request.send().await.unwrap(); + let after = CellAuthority::new(fixture.layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .incarnation; + assert_eq!( + before, after, + "restoration must retain the original coordinator generation" + ); + let item = fixture + .sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("same-cold".into())) + .consistent_read(true) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::N("7".into())); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn pending_cold_admission_reserves_the_last_slot_and_cancellation_releases_it() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 5, true).await; + assert_eq!(fixture.node.runtime().stats().active_cells(), 4); + let first = b"slot-first".to_vec(); + let first_target = beyonddb::coordinator_target(ACCOUNT, &first).unwrap(); + let second = (0..100) + .map(|index| format!("slot-second-{index}").into_bytes()) + .find(|key| { + beyonddb::coordinator_target(ACCOUNT, key) + .unwrap() + .cell_id() + .as_bytes()[0] + % 64 + != first_target.cell_id().as_bytes()[0] % 64 + }) + .unwrap(); + let second_target = beyonddb::coordinator_target(ACCOUNT, &second).unwrap(); + let entered = Arc::new(tokio::sync::Semaphore::new(0)); + let release = Arc::new(tokio::sync::Semaphore::new(0)); + *store.creation_gate.lock().unwrap() = Some(super::forward_cache::CreationGate { + paths: [&first_target, &second_target] + .into_iter() + .map(|target| fixture.layout.control_path(target.cell_id().as_bytes())) + .collect(), + entered: entered.clone(), + release: release.clone(), + }); + let spawn = |key: Vec| { + let provisioner = fixture.provisioner.clone(); + let client = fixture.client.clone(); + tokio::spawn(async move { provisioner.ensure(&client, ACCOUNT, &key).await }) + }; + let first_job = spawn(first); + tokio::time::timeout(std::time::Duration::from_secs(2), entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let second_job = spawn(second); + let blocked = tokio::time::timeout(std::time::Duration::from_millis(150), async { + entered.acquire().await.unwrap().forget(); + }) + .await + .is_err(); + first_job.abort(); + assert!(first_job.await.unwrap_err().is_cancelled()); + if blocked { + tokio::time::timeout(std::time::Duration::from_secs(2), entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + } + release.add_permits(2); + tokio::time::timeout(std::time::Duration::from_secs(10), second_job) + .await + .unwrap() + .unwrap() + .unwrap(); + assert!( + blocked, + "a second cold claim bypassed the pending last-slot reservation" + ); + let authority = CellAuthority::new(fixture.layout.clone()); + assert!( + authority + .load(first_target.cell_id()) + .await + .unwrap() + .is_none() + ); + assert!( + authority + .load(second_target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .root + .is_some() + ); + assert!(fixture.node.runtime().stats().active_cells() <= 5); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn distinct_cold_sdk_coordinators_overlap_without_losing_durable_registration() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 16, true).await; + let first = b"cold-parallel-0".to_vec(); + let first_target = beyonddb::coordinator_target(ACCOUNT, &first).unwrap(); + let (second, second_target) = (1..100) + .map(|index| { + let key = format!("cold-parallel-{index}").into_bytes(); + let target = beyonddb::coordinator_target(ACCOUNT, &key).unwrap(); + (key, target) + }) + .find(|(_, target)| { + target.cell_id().as_bytes()[0] % 64 != first_target.cell_id().as_bytes()[0] % 64 + }) + .unwrap(); + let entered = Arc::new(tokio::sync::Semaphore::new(0)); + let release = Arc::new(tokio::sync::Semaphore::new(0)); + *store.creation_gate.lock().unwrap() = Some(super::forward_cache::CreationGate { + paths: [&first_target, &second_target] + .into_iter() + .map(|target| fixture.layout.control_path(target.cell_id().as_bytes())) + .collect(), + entered: entered.clone(), + release: release.clone(), + }); + let mut jobs = Vec::new(); + for (index, token) in [first.clone(), second.clone()].into_iter().enumerate() { + let sdk = fixture.sdk.clone(); + jobs.push(tokio::spawn(async move { + sdk.transact_write_items() + .client_request_token(String::from_utf8(token).unwrap()) + .transact_items( + aws_sdk_dynamodb::types::TransactWriteItem::builder() + .put( + aws_sdk_dynamodb::types::Put::builder() + .table_name("Residency") + .item("id", AwsAttributeValue::S(format!("cold-{index}"))) + .item("value", AwsAttributeValue::N(index.to_string())) + .build() + .unwrap(), + ) + .build(), + ) + .send() + .await + })); + } + let overlap = tokio::time::timeout(std::time::Duration::from_secs(1), async { + entered.acquire_many(2).await.unwrap().forget(); + }) + .await; + release.add_permits(2); + for job in jobs { + tokio::time::timeout(std::time::Duration::from_secs(10), job) + .await + .unwrap() + .unwrap() + .unwrap(); + } + assert!( + overlap.is_ok(), + "independent cold SDK coordinator creates were serialized" + ); + for (key, target) in [(&first, first_target), (&second, second_target)] { + let registered = fixture + .client + .query::( + &account_target(ACCOUNT).unwrap(), + None, + Json(RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: u32::from_be_bytes(target.partition().try_into().unwrap()), + }), + ) + .await + .unwrap(); + assert!(registered.output.0); + let control = CellAuthority::new(fixture.layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(control.value().root.is_some()); + fixture + .provisioner + .ensure(&fixture.client, ACCOUNT, key) + .await + .unwrap(); + } + fixture.data[0].0.drain().await.unwrap(); + for index in 0..2 { + let item = fixture + .sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S(format!("cold-{index}"))) + .consistent_read(true) + .send() + .await + .unwrap(); + assert_eq!( + item.item.unwrap()["value"], + AwsAttributeValue::N(index.to_string()) + ); + } + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn registered_resident_coordinator_admission_skips_provider_io_but_restores_after_drain() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let key = b"resident-admission"; + let coordinator = beyonddb::coordinator_target(ACCOUNT, key).unwrap(); + let handle = fixture + .provisioner + .admit_coordinator(ACCOUNT, key) + .await + .unwrap(); + *store.path.lock().unwrap() = Some( + fixture + .layout + .control_path(coordinator.cell_id().as_bytes()), + ); + store.reads.store(0, Ordering::SeqCst); + store.all_reads.store(0, Ordering::SeqCst); + let start = std::time::Instant::now(); + for _ in 0..4 { + fixture + .provisioner + .ensure(&fixture.client, ACCOUNT, key) + .await + .unwrap(); + } + let reads = store.reads.load(Ordering::SeqCst); + println!( + "four registered resident admissions: {:?}; coordinator authority reads={reads}", + start.elapsed() + ); + assert_eq!( + reads, 0, + "a registered resident shard should not need provider reads on every admission" + ); + assert_eq!( + store.all_reads.load(Ordering::SeqCst), + 0, + "resident admission must not read any provider object" + ); + // A new provisioner has no registration receipt, even on the same runtime. + let fresh = CellInitialPartitionProvisioner::new( + fixture.node.runtime(), + fixture.application.clone(), + fixture.layout.clone(), + fixture.session, + fixture.endpoint.clone(), + fixture._files.path().join("fresh-admission"), + ) + .unwrap(); + store.reads.store(0, Ordering::SeqCst); + fresh.ensure(&fixture.client, ACCOUNT, key).await.unwrap(); + assert!( + store.reads.load(Ordering::SeqCst) > 0, + "a new cache must establish registration from durable state" + ); + // Losing account residency invalidates the original admission shortcut. + fixture + .provisioner + .admit_account(ACCOUNT) + .await + .unwrap() + .drain() + .await + .unwrap(); + store.reads.store(0, Ordering::SeqCst); + fixture + .provisioner + .ensure(&fixture.client, ACCOUNT, key) + .await + .unwrap(); + assert!( + store.reads.load(Ordering::SeqCst) > 0, + "a drained account must re-enter canonical admission" + ); + handle.drain().await.unwrap(); + store.reads.store(0, Ordering::SeqCst); + fixture + .provisioner + .ensure(&fixture.client, ACCOUNT, key) + .await + .unwrap(); + assert!( + store.reads.load(Ordering::SeqCst) > 0, + "a drained coordinator must re-enter canonical admission" + ); + let proof = CellCatalog::new(fixture.layout.clone(), coordinator.tenant()) + .lookup(coordinator.cell_id()) + .await + .unwrap() + .unwrap(); + let control = CellAuthority::new(fixture.layout.clone()) + .load(coordinator.cell_id()) + .await + .unwrap() + .unwrap(); + assert!( + fixture + .node + .runtime() + .local_handle(proof, &control) + .await + .unwrap() + .is_some() + ); + *store.path.lock().unwrap() = None; + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn resident_coordinator_without_acknowledged_registration_is_not_cached() { + let fixture = Fixture::with_capacity(1, 8).await; + let key = b"unregistered-resident"; + let coordinator = beyonddb::coordinator_target(ACCOUNT, key).unwrap(); + fixture + .provisioner + .recover_owned_coordinator(ACCOUNT, key, &fixture.directory) + .await + .unwrap(); + let input = RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: u32::from_be_bytes(coordinator.partition().try_into().unwrap()), + }; + let account = account_target(ACCOUNT).unwrap(); + assert!( + !fixture + .client + .query::(&account, None, Json(input.clone())) + .await + .unwrap() + .output + .0 + ); + fixture + .provisioner + .ensure(&fixture.client, ACCOUNT, key) + .await + .unwrap(); + assert!( + fixture + .client + .query::(&account, None, Json(input)) + .await + .unwrap() + .output + .0 + ); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn missing_coordinator_authority_is_not_read_twice_before_creation_cas() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let key = b"fresh-admission-absence"; + let coordinator = beyonddb::coordinator_target(ACCOUNT, key).unwrap(); + *store.path.lock().unwrap() = Some( + fixture + .layout + .control_path(coordinator.cell_id().as_bytes()), + ); + store.reads.store(0, Ordering::SeqCst); + fixture + .provisioner + .ensure(&fixture.client, ACCOUNT, key) + .await + .unwrap(); + let reads = store.reads.load(Ordering::SeqCst); + println!("first coordinator admission authority reads={reads}"); + assert_eq!( + reads, 1, + "the conditional creation must revalidate observed absence without another GET" + ); + let account = account_target(ACCOUNT).unwrap(); + let input = RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: u32::from_be_bytes(coordinator.partition().try_into().unwrap()), + }; + assert!( + fixture + .client + .query::(&account, None, Json(input)) + .await + .unwrap() + .output + .0 + ); + *store.path.lock().unwrap() = None; + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn missing_coordinator_observation_never_overwrites_a_competing_owner() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let remote = super::provisioning::Remote::new(&fixture).await; + let key = b"fresh-admission-race"; + let coordinator = beyonddb::coordinator_target(ACCOUNT, key).unwrap(); + let entered = Arc::new(tokio::sync::Semaphore::new(0)); + let release = Arc::new(tokio::sync::Semaphore::new(0)); + *store.creation_gate.lock().unwrap() = Some(super::forward_cache::CreationGate { + paths: [fixture + .layout + .control_path(coordinator.cell_id().as_bytes())] + .into_iter() + .collect(), + entered: entered.clone(), + release: release.clone(), + }); + let local = { + let provisioner = fixture.provisioner.clone(); + let client = fixture.client.clone(); + tokio::spawn(async move { provisioner.ensure(&client, ACCOUNT, key).await }) + }; + tokio::time::timeout(std::time::Duration::from_secs(2), entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + // The local request has observed absence and is waiting at creation CAS. + // Let a real peer create and publish the same Cell before releasing it. + *store.creation_gate.lock().unwrap() = None; + let foreign = remote + .provisioner + .recover_owned_coordinator(ACCOUNT, key, &fixture.directory) + .await + .unwrap(); + release.add_permits(1); + assert!( + tokio::time::timeout(std::time::Duration::from_secs(10), local) + .await + .unwrap() + .unwrap() + .is_err() + ); + let assert_foreign_owner = || async { + let current = CellAuthority::new(fixture.layout.clone()) + .load(coordinator.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value().incarnation, foreign.incarnation()); + assert_eq!( + current.value().owner.as_ref().unwrap().session, + remote.session + ); + assert!(current.value().root.is_some()); + }; + assert_foreign_owner().await; + // A retry discovers the published peer root and can register it locally. + fixture + .provisioner + .ensure(&fixture.client, ACCOUNT, key) + .await + .unwrap(); + assert_foreign_owner().await; + let account = account_target(ACCOUNT).unwrap(); + assert!( + fixture + .client + .query::( + &account, + None, + Json(RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: u32::from_be_bytes(coordinator.partition().try_into().unwrap()), + }), + ) + .await + .unwrap() + .output + .0 + ); + remote.shutdown().await; + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn cold_sdk_coordinator_overlaps_catalog_publication_with_authority_lookup() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 16, true).await; + let token = "cold-catalog-overlap"; + let account = account_target(ACCOUNT).unwrap(); + let target = beyonddb::coordinator_target(ACCOUNT, token.as_bytes()).unwrap(); + let entered = Arc::new(tokio::sync::Semaphore::new(0)); + let release = Arc::new(tokio::sync::Semaphore::new(0)); + let catalog = super::forward_cache::CreationGate { + paths: [fixture + .layout + .catalog_head_path(account.tenant().as_bytes(), target.cell_id().as_bytes()[0])] + .into_iter() + .collect(), + entered: entered.clone(), + release: release.clone(), + }; + *store.creation_gate.lock().unwrap() = Some(catalog.clone()); + *store.publication_gate.lock().unwrap() = Some(catalog); + *store.read_gate.lock().unwrap() = Some(super::forward_cache::CreationGate { + paths: [fixture.layout.control_path(target.cell_id().as_bytes())] + .into_iter() + .collect(), + entered: entered.clone(), + release: release.clone(), + }); + let request = fixture + .sdk + .transact_write_items() + .client_request_token(token) + .transact_items( + aws_sdk_dynamodb::types::TransactWriteItem::builder() + .put( + aws_sdk_dynamodb::types::Put::builder() + .table_name("Residency") + .item("id", AwsAttributeValue::S("cold-catalog-overlap".into())) + .item("value", AwsAttributeValue::N("7".into())) + .build() + .unwrap(), + ) + .build(), + ); + let job = tokio::spawn(request.clone().send()); + let overlap = tokio::time::timeout(std::time::Duration::from_secs(2), async { + entered.acquire_many(2).await.unwrap().forget(); + }) + .await + .is_ok(); + // Release every gate before collecting the SDK result, including on the + // serial baseline. The assertion checks dependencies, not request timeout. + *store.creation_gate.lock().unwrap() = None; + *store.publication_gate.lock().unwrap() = None; + *store.read_gate.lock().unwrap() = None; + release.add_permits(2); + tokio::time::timeout(std::time::Duration::from_secs(10), job) + .await + .unwrap() + .unwrap() + .unwrap(); + let registered = fixture + .client + .query::( + &account, + None, + Json(RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: u32::from_be_bytes(target.partition().try_into().unwrap()), + }), + ) + .await + .unwrap() + .output + .0; + assert!(registered, "SDK success requires durable account discovery"); + let control = CellAuthority::new(fixture.layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(control.value().root.is_some()); + let incarnation = control.value().incarnation; + fixture + .provisioner + .admit_coordinator(ACCOUNT, token.as_bytes()) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_coordinator(ACCOUNT, token.as_bytes()) + .await + .unwrap(); + request.send().await.unwrap(); + let restored = CellAuthority::new(fixture.layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(restored.value().incarnation, incarnation); + let item = fixture + .sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("cold-catalog-overlap".into())) + .consistent_read(true) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::N("7".into())); + fixture.shutdown().await; + assert!( + overlap, + "cold coordinator catalog publication waited for the independent authority lookup" + ); +} diff --git a/tests/peer_network/residency/coordinator_discovery.rs b/tests/peer_network/residency/coordinator_discovery.rs new file mode 100644 index 0000000..f9d55f0 --- /dev/null +++ b/tests/peer_network/residency/coordinator_discovery.rs @@ -0,0 +1,184 @@ +//! Discovery must recover the account index before abandoned coordinator shards. + +use super::provisioning::{Remote, table_id, wait_for_expiry}; +use super::*; +use beyonddb::{ + BeginCrossCellTransaction, BeginCrossCellTransactionInput, CoordinatorDecision, + CoordinatorParticipant, CoordinatorParticipantTarget, GetItemInput, + IndexedTransactionOperation, ReadCrossCellTransaction, ReadCrossCellTransactionInput, + TransactionOperation, +}; +use cellule_runtime::cell::catalog::CellCatalog; +use std::time::Duration; + +const ACCOUNT: &str = "123456789012"; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn recovery_discovers_new_coordinator_after_its_account_owner_expires() { + let fixture = Fixture::with_capacity(2, 16).await; + let remote = Remote::new(&fixture).await; + let account = account_target(ACCOUNT).unwrap(); + let table = table_id(&fixture, "Residency").await; + let route = crate::single_leaf_route(&fixture.client, &account, &table) + .await + .unwrap(); + fixture + .provisioner + .admit_account(ACCOUNT) + .await + .unwrap() + .drain() + .await + .unwrap(); + remote.provisioner.admit_account(ACCOUNT).await.unwrap(); + // Install before registration. This worker has never admitted the new shard, + // so its only durable discovery path is the account's coordinator index. + fixture + .provisioner + .install_transaction_recovery_loop( + &fixture.tasks, + beyonddb::CellStorage::new(fixture.client.clone(), "us-east-1"), + fixture.directory.clone(), + vec![ACCOUNT.into()], + ) + .unwrap(); + let transaction_id = [174; 16]; + let coordinator = beyonddb::coordinator_target(ACCOUNT, &transaction_id).unwrap(); + remote + .provisioner + .admit_coordinator(ACCOUNT, &transaction_id) + .await + .unwrap(); + let mut participants = route + .partitions + .iter() + .enumerate() + .map(|(index, partition)| { + let target = beyonddb::data_target(ACCOUNT, &table, &partition.partition_id).unwrap(); + let (_, item) = fixture + .data + .iter() + .find(|(handle, _)| handle.cell_id() == target.cell_id()) + .unwrap(); + CoordinatorParticipant { + target: CoordinatorParticipantTarget::Data { + table_id: table.clone(), + partition_id: partition.partition_id, + epoch: partition.epoch, + }, + operations: vec![IndexedTransactionOperation { + index: u8::try_from(index).unwrap(), + operation: TransactionOperation::Read(GetItemInput { + table_name: "Residency".into(), + table_id: table.clone(), + key: Item::from([( + "id".into(), + AttributeValue::S(item["id"].as_s().unwrap().clone()), + )]), + }), + }], + } + }) + .collect::>(); + participants.sort_by_key(|participant| match &participant.target { + CoordinatorParticipantTarget::Data { + table_id, + partition_id, + .. + } => *beyonddb::data_target(ACCOUNT, table_id, partition_id) + .unwrap() + .cell_id() + .as_bytes(), + _ => unreachable!(), + }); + let now = now_ms(); + let identity = cellule_runtime::MutationIdentity { + request_id: cellule_runtime::identity::RequestId::from_bytes( + *uuid::Uuid::now_v7().as_bytes(), + ), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }; + transaction_command!( + remote.client(&fixture), + BeginCrossCellTransaction, + &coordinator, + identity, + Json(BeginCrossCellTransactionInput { + account_id: ACCOUNT.into(), + transaction_id, + token: None, + participants, + }), + ) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(750)).await; + let authority = CellAuthority::new(fixture.layout.clone()); + for target in [&account, &coordinator] { + let current = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!( + current.value().owner.as_ref().unwrap().session, + remote.session, + "discovery must preserve a live remote owner" + ); + } + remote.stop_listener(); + remote.lease.cancel(); + wait_for_expiry(&fixture, remote.session).await; + let proof = CellCatalog::new(fixture.layout.clone(), coordinator.tenant()) + .lookup(coordinator.cell_id()) + .await + .unwrap() + .unwrap(); + let recovered = tokio::time::timeout(Duration::from_secs(15), async { + loop { + let current = authority + .load(coordinator.cell_id()) + .await + .unwrap() + .unwrap(); + // Observe locally: a public/routed query could restore the shard and + // hide a worker discovery failure. The observer never admits work. + if let Some(handle) = fixture + .node + .runtime() + .local_handle(proof.clone(), ¤t) + .await + .unwrap() + { + let status = CellClient::local(fixture.application.registry(), handle) + .query::( + &coordinator, + None, + Json(ReadCrossCellTransactionInput { + account_id: ACCOUNT.into(), + transaction_id, + routing_key: transaction_id.to_vec(), + }), + ) + .await + .unwrap() + .output + .0 + .unwrap(); + if status.resolved_count == 2 { + assert_eq!(status.decision, CoordinatorDecision::Commit); + break; + } + } + tokio::time::sleep(Duration::from_millis(25)).await; + } + }) + .await; + // This expired owner still has an abandoned BEGIN. Its cleanup may report + // the exact fence that prevents a stale writer from publishing on shutdown. + let stopped = remote.node.shutdown().await; + assert!( + matches!(stopped, Ok(()) | Err(cellule_runtime::Error::Fenced)), + "unexpected lost-owner shutdown error: {stopped:?}" + ); + fixture.shutdown().await; + recovered + .expect("worker must recover account discovery and resolve the new abandoned coordinator"); +} diff --git a/tests/peer_network/residency/coordinator_registration.rs b/tests/peer_network/residency/coordinator_registration.rs new file mode 100644 index 0000000..278d895 --- /dev/null +++ b/tests/peer_network/residency/coordinator_registration.rs @@ -0,0 +1,278 @@ +use super::*; +use beyonddb::{ + CoordinatorProvisioner, ReadCoordinatorRegistration, RegisterCoordinatorShardInput, + RegisterCoordinatorShards, RegisterCoordinatorShardsInput, +}; +use cellule_runtime::codec::{BoundedEncoder, WireValue}; + +const ACCOUNT: &str = "123456789012"; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn cancellation_during_registration_publication_keeps_discovery_and_waits_for_ack() { + let store = Arc::new(super::forward_cache::CountedAuthority::default()); + // Use canonical routing for the post-drain queries. A direct cached Cell + // client can return CellDraining until its short-lived handle expires. + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 12, false).await; + let account = account_target(ACCOUNT).unwrap(); + let first_key = b"canceled-registration"; + let first_target = beyonddb::coordinator_target(ACCOUNT, first_key).unwrap(); + let second_key = (0..100) + .map(|index| format!("healthy-registration-{index}").into_bytes()) + .find(|key| beyonddb::coordinator_target(ACCOUNT, key).unwrap() != first_target) + .unwrap(); + for key in [first_key.as_slice(), second_key.as_slice()] { + fixture + .provisioner + .recover_owned_coordinator(ACCOUNT, key, &fixture.directory) + .await + .unwrap(); + } + let entered = Arc::new(tokio::sync::Semaphore::new(0)); + let release = Arc::new(tokio::sync::Semaphore::new(0)); + *store.publication_gate.lock().unwrap() = Some(super::forward_cache::CreationGate { + paths: [fixture.layout.control_path(account.cell_id().as_bytes())] + .into_iter() + .collect(), + entered: entered.clone(), + release: release.clone(), + }); + let spawn = |key: Vec| { + let provisioner = fixture.provisioner.clone(); + let client = fixture.client.clone(); + tokio::spawn(async move { provisioner.ensure(&client, ACCOUNT, &key).await }) + }; + let first = spawn(first_key.to_vec()); + tokio::time::timeout(std::time::Duration::from_secs(2), entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + assert!( + !first.is_finished(), + "admission cannot acknowledge an unpublished registration" + ); + first.abort(); + assert!(first.await.unwrap_err().is_cancelled()); + let second = spawn(second_key.clone()); + *store.publication_gate.lock().unwrap() = None; + release.add_permits(1); + tokio::time::timeout(std::time::Duration::from_secs(10), second) + .await + .unwrap() + .unwrap() + .unwrap(); + fixture + .provisioner + .admit_account(ACCOUNT) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture.provisioner.admit_account(ACCOUNT).await.unwrap(); + for key in [first_key.as_slice(), second_key.as_slice()] { + let target = beyonddb::coordinator_target(ACCOUNT, key).unwrap(); + assert!( + fixture + .client + .query::( + &account, + None, + Json(RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: u32::from_be_bytes(target.partition().try_into().unwrap()), + }), + ) + .await + .unwrap() + .output + .0 + ); + } + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn concurrent_coordinator_admissions_share_durable_account_registration() { + let fixture = Fixture::with_capacity(1, 16).await; + let account = account_target(ACCOUNT).unwrap(); + let mut cells = std::collections::HashSet::new(); + let keys: Vec<_> = (0..100) + .map(|index| format!("registration-batch-{index}").into_bytes()) + .filter(|key| { + cells.insert( + beyonddb::coordinator_target(ACCOUNT, key) + .unwrap() + .cell_id(), + ) + }) + .take(4) + .collect(); + // Give admission published but undiscoverable shards. This isolates the + // account registration cost from independent coordinator bootstrap I/O. + for key in &keys { + fixture + .provisioner + .recover_owned_coordinator(ACCOUNT, key, &fixture.directory) + .await + .unwrap(); + } + let sequence = || async { + CellAuthority::new(fixture.layout.clone()) + .load(account.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .root + .as_ref() + .unwrap() + .commit_sequence + }; + let before = sequence().await; + let barrier = Arc::new(tokio::sync::Barrier::new(keys.len())); + let mut jobs = Vec::new(); + for key in keys.clone() { + let provisioner = fixture.provisioner.clone(); + let client = fixture.client.clone(); + let barrier = barrier.clone(); + jobs.push(tokio::spawn(async move { + barrier.wait().await; + provisioner.ensure(&client, ACCOUNT, &key).await + })); + } + for job in jobs { + tokio::time::timeout(std::time::Duration::from_secs(10), job) + .await + .unwrap() + .unwrap() + .unwrap(); + } + let commits = sequence().await - before; + println!("four concurrent coordinator registrations account commits={commits}"); + assert!( + commits > 0 && commits < keys.len() as u64, + "concurrent admissions should publish fewer registration commands than requests" + ); + fixture + .provisioner + .admit_account(ACCOUNT) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture.provisioner.admit_account(ACCOUNT).await.unwrap(); + for key in &keys { + let target = beyonddb::coordinator_target(ACCOUNT, key).unwrap(); + assert!( + fixture + .client + .query::( + &account, + None, + Json(RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: u32::from_be_bytes(target.partition().try_into().unwrap()), + }), + ) + .await + .unwrap() + .output + .0, + "acknowledged registration must survive owner restoration" + ); + } + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn coordinator_registration_batch_enforces_scope_and_limits_without_partial_rows() { + let fixture = Fixture::with_capacity(1, 12).await; + let account = account_target(ACCOUNT).unwrap(); + for (account_id, shards) in [ + (ACCOUNT, vec![]), + (ACCOUNT, (128..193).collect()), + (ACCOUNT, vec![1234, 4096]), + ("other-account", vec![1234]), + ] { + assert!( + fixture + .client + .command::( + &account, + mutation(), + Json(RegisterCoordinatorShardsInput { + account_id: account_id.into(), + shards, + }), + ) + .await + .is_err() + ); + } + assert!( + !fixture + .client + .query::( + &account, + None, + Json(RegisterCoordinatorShardInput { + account_id: ACCOUNT.into(), + shard: 1234, + }), + ) + .await + .unwrap() + .output + .0 + ); + + // Exercise the largest supported account encoding and batch on real Cells. + let maximal_account = "\0".repeat(128); + let input = RegisterCoordinatorShardsInput { + account_id: maximal_account.clone(), + shards: vec![4095; 64], + }; + let mut encoder = BoundedEncoder::new(4096).unwrap(); + Json(input.clone()).encode(&mut encoder).unwrap(); + fixture + .provisioner + .admit_account(&maximal_account) + .await + .unwrap(); + let target = account_target(&maximal_account).unwrap(); + fixture + .client + .command::(&target, mutation(), Json(input)) + .await + .unwrap(); + assert!( + fixture + .client + .query::( + &target, + None, + Json(RegisterCoordinatorShardInput { + account_id: maximal_account, + shard: 4095, + }), + ) + .await + .unwrap() + .output + .0 + ); + fixture.shutdown().await; +} + +fn mutation() -> cellule_runtime::MutationIdentity { + let now = now_ms(); + cellule_runtime::MutationIdentity { + request_id: cellule_runtime::identity::RequestId::from_bytes( + *uuid::Uuid::now_v7().as_bytes(), + ), + issued_at_ms: now, + expires_at_ms: now + 60_000, + } +} diff --git a/tests/peer_network/residency/credential_failover.rs b/tests/peer_network/residency/credential_failover.rs new file mode 100644 index 0000000..f257942 --- /dev/null +++ b/tests/peer_network/residency/credential_failover.rs @@ -0,0 +1,130 @@ +use super::*; +use cellule_runtime::cell::catalog::CellCatalog; + +const ACCESS_KEY: &str = "AKIAIOSFODNN7EXAMPLE"; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn signed_sdk_authentication_recovers_expired_published_credential_owner() { + let mut recovered = Vec::new(); + for enabled in [false, true] { + let fixture = + Fixture::with_store_capacity_and_peer_cache(1, Arc::new(InMemory::new()), 8, enabled) + .await; + let remote = provisioning::Remote::new(&fixture).await; + let target = beyonddb::credential_target(ACCESS_KEY).unwrap(); + let original = fixture + .provisioner + .admit_credential(ACCESS_KEY) + .await + .unwrap(); + original.drain().await.unwrap(); + let serving = remote + .provisioner + .admit_credential(ACCESS_KEY) + .await + .unwrap(); + let before = serving.owner_fence(); + let sdk = provisioning::sdk_without_retries(&fixture); + let expected = fixture.data[0].1.clone(); + let live_read = sdk + .get_item() + .table_name("Residency") + .key("id", expected["id"].clone()) + .consistent_read(true) + .send() + .await + .unwrap(); + assert_eq!(live_read.item, Some(expected.clone())); + let unknown = (0..256) + .map(|index| format!("AKIAUNKNOWN{index:08}")) + .find(|key| beyonddb::credential_target(key).unwrap().cell_id() != target.cell_id()) + .unwrap(); + let unknown_target = beyonddb::credential_target(&unknown).unwrap(); + let unknown_sdk = aws_sdk_dynamodb::Client::from_conf( + sdk.config() + .to_builder() + .credentials_provider(Credentials::new( + unknown, + "unrecognized-secret", + None, + None, + "unknown-key-test", + )) + .build(), + ); + let unknown_error = unknown_sdk + .get_item() + .table_name("Residency") + .key("id", expected["id"].clone()) + .send() + .await + .unwrap_err(); + assert_eq!( + unknown_error.as_service_error().unwrap().code(), + Some("UnrecognizedClientException") + ); + remote.stop_listener(); + assert!( + sdk.get_item() + .table_name("Residency") + .key("id", expected["id"].clone()) + .consistent_read(true) + .send() + .await + .is_err() + ); + let authority = CellAuthority::new(fixture.layout.clone()); + let live = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert!( + fixture + .directory + .is_live(remote.session, now_ms()) + .await + .unwrap() + ); + assert_eq!(live.value().incarnation, before.incarnation); + assert_eq!(live.value().epoch, before.epoch); + assert_eq!(live.value().owner.as_ref().unwrap().session, remote.session); + remote.lease.cancel(); + provisioning::wait_for_expiry(&fixture, remote.session).await; + let item = sdk + .get_item() + .table_name("Residency") + .key("id", expected["id"].clone()) + .consistent_read(true) + .send() + .await; + let after = authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + let unknown_control = authority.load(unknown_target.cell_id()).await.unwrap(); + let unknown_catalog = CellCatalog::new(fixture.layout.clone(), unknown_target.tenant()) + .lookup(unknown_target.cell_id()) + .await + .unwrap(); + let session = fixture.session; + assert!(matches!( + remote.node.shutdown().await, + Ok(()) | Err(cellule_runtime::Error::Fenced) + )); + fixture.shutdown().await; + assert!(unknown_control.is_none()); + assert!(unknown_catalog.is_none()); + recovered.push((enabled, item, expected, after, before, session)); + } + for (enabled, item, expected, after, before, session) in recovered { + assert_eq!( + item.unwrap_or_else(|error| panic!("credential recovery cache={enabled}: {error:?}")) + .item, + Some(expected) + ); + assert_eq!(after.incarnation, before.incarnation); + assert!(after.epoch > before.epoch); + assert_eq!(after.owner.as_ref().unwrap().session, session); + assert!(after.root.is_some()); + } +} diff --git a/tests/peer_network/residency/follower_durability.rs b/tests/peer_network/residency/follower_durability.rs new file mode 100644 index 0000000..5471b8c --- /dev/null +++ b/tests/peer_network/residency/follower_durability.rs @@ -0,0 +1,483 @@ +use super::*; + +use std::sync::atomic::{AtomicBool, Ordering}; +use std::{fmt, sync::Mutex, time::Duration}; + +use async_trait::async_trait; +use beyonddb::{PeerNodeDurabilityProvider, PeerNodeLogTransport, RuntimeMetrics}; +use cellule_host::{FOLLOWER_STORE_COMPONENT, NodeDurabilitySupervisorConfig}; +use cellule_runtime::node::{NODE_LOG_PROTOCOL_VERSION, NodeCapacity}; +use cellule_runtime::{NodeLeaseGuard, follower::FollowerStore, ltx::Limits}; +use futures_util::stream::BoxStream; +use object_store::{ + CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, + PutMultipartOptions, PutOptions, PutPayload, PutResult, path::Path, +}; + +// Only immutable objects for the selected Cell are withheld. Catalogs, leases, +// log activation, and other Cells remain available throughout the experiment. +#[derive(Debug)] +struct WithheldPublication { + inner: InMemory, + cell: Mutex>, + blocked: CancellationToken, + release: CancellationToken, +} + +impl WithheldPublication { + fn new() -> Self { + Self { + inner: InMemory::new(), + cell: Mutex::new(None), + blocked: CancellationToken::new(), + release: CancellationToken::new(), + } + } + + fn withhold(&self, cell: &[u8; 32]) { + let hex = cell + .iter() + .map(|byte| format!("{byte:02x}")) + .collect::(); + *self.cell.lock().unwrap() = Some(format!("/cells/{hex}/inc/")); + } + + async fn wait(&self, path: &Path) { + let blocked = self.cell.lock().unwrap().as_ref().is_some_and(|cell| { + path.as_ref().contains(cell) && path.as_ref().contains("/objects/") + }); + if blocked { + self.blocked.cancel(); + self.release.cancelled().await; + } + } +} + +impl fmt::Display for WithheldPublication { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str("WithheldPublication") + } +} + +#[async_trait] +impl ObjectStore for WithheldPublication { + async fn put_opts( + &self, + path: &Path, + payload: PutPayload, + opts: PutOptions, + ) -> object_store::Result { + self.wait(path).await; + self.inner.put_opts(path, payload, opts).await + } + async fn put_multipart_opts( + &self, + path: &Path, + opts: PutMultipartOptions, + ) -> object_store::Result> { + self.wait(path).await; + self.inner.put_multipart_opts(path, opts).await + } + async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { + self.inner.get_opts(path, opts).await + } + fn delete_stream( + &self, + paths: BoxStream<'static, object_store::Result>, + ) -> BoxStream<'static, object_store::Result> { + self.inner.delete_stream(paths) + } + fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { + self.inner.list(prefix) + } + async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { + self.inner.list_with_delimiter(prefix).await + } + async fn copy_opts( + &self, + from: &Path, + to: &Path, + opts: CopyOptions, + ) -> object_store::Result<()> { + self.wait(to).await; + self.inner.copy_opts(from, to, opts).await + } +} + +struct LogNode { + node: CellNode, + metrics: Arc, + _tasks: Arc, + provisioner: Arc, + guard: NodeLeaseGuard, + session: SessionId, + id: NodeId, + crash: CancellationToken, + server: tokio::task::JoinHandle<()>, +} + +impl LogNode { + async fn new(fixture: &Fixture, byte: u8, follower: bool, durability: bool) -> Self { + let root = fixture._files.path(); + let (cert, key) = peer_tls_files( + root, + &format!("log-{byte}"), + &root.join("ca.crt"), + &root.join("ca.key"), + ); + let tls = LoadedPeerTls::load(&cert, &key, &root.join("ca.crt"), "localhost").unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("https://{}", listener.local_addr().unwrap()); + let session = SessionId::from_bytes([byte; 16]); + let id = NodeId::from_bytes([byte; 16]); + let mut builder = CellNodeBuilder::new(fixture.application.clone()) + .with_runtime(SqlWorkerPool::new(2, 16).unwrap(), 64 << 20) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(1 << 30))) + .with_session(session); + if follower { + builder = builder.with_follower_store( + root.join(format!("follower-{byte}")), + Limits::default(), + DiskBudget::new(1 << 30), + ); + } + let node = builder.build().unwrap(); + let metrics = Arc::new(RuntimeMetrics::default()); + node.install_telemetry(metrics.clone()).unwrap(); + let crash = CancellationToken::new(); + let shutdown = CancellationToken::new(); + let tasks = node + .install_task_group(crash.child_token(), shutdown.clone()) + .unwrap(); + let signing = tls.signing_key().clone(); + let certificate = tls.certificate(); + let fleet = fixture.directory.fleet(); + let release = fixture.application.registry().release_digest(); + let modules = fixture.application.registry().module_digests(); + let published = NodeLeasePublisher::new(fixture.directory.clone(), { + let endpoint = endpoint.clone(); + move |now, expires| { + NodeAdvertisement::sign( + id, + session, + endpoint.clone(), + fleet, + certificate, + Digest::from_bytes([90; 32]), + release, + &signing, + 1, + now, + expires, + modules.clone(), + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + log_protocol: NODE_LOG_PROTOCOL_VERSION, + follower_free_bytes: if follower { 1 << 30 } else { 0 }, + free_memory_bytes: 64 << 20, + free_disk_bytes: 1 << 30, + job_credits: 16, + ..NodeCapacity::default() + }, + ) + } + }) + .publish() + .await + .unwrap(); + let guard = published.guard(); + let authority = published.log_authority(); + node.install_node_lease_for_startup(guard.clone()).unwrap(); + tasks + .spawn_lease_maintenance({ + let crash = crash.clone(); + async move { + tokio::select! { + () = crash.cancelled() => Ok(()), + result = published.run(&shutdown) => result, + } + } + }) + .unwrap(); + let transport = Arc::new(PeerNodeLogTransport::new( + fixture.directory.clone(), + tls.client_identity(), + session, + id, + guard.clone(), + )); + let mut peers = BeyonddbPeers::new( + &node, + fixture.layout.clone(), + fixture.directory.clone(), + session, + &tls, + ) + .unwrap(); + if let Some(store) = node.owned_component::(FOLLOWER_STORE_COMPONENT) { + peers = peers.with_follower_store(id, store, guard.clone()); + } + let peers = Arc::new(peers); + let provisioner = Arc::new( + CellInitialPartitionProvisioner::new( + node.runtime(), + fixture.application.clone(), + fixture.layout.clone(), + session, + endpoint, + root.join(format!("log-data-{byte}")), + ) + .unwrap() + .with_peers(peers.clone()) + .with_node_log_recovery(transport.clone()) + .unwrap(), + ); + let router = peers.router(provisioner.clone()); + let server = tokio::spawn(async move { + axum::serve( + tls.listener(listener), + router.into_make_service_with_connect_info::(), + ) + .await + .unwrap(); + }); + if durability { + let ready = Arc::new(AtomicBool::new(false)); + let provider = Arc::new( + PeerNodeDurabilityProvider::new( + authority, + transport.as_ref().clone(), + session, + id, + guard.clone(), + node.runtime().telemetry_handle(), + ) + .unwrap() + .with_recruitment_gate(ready.clone()), + ); + let recruited = cellule_host::NodeDurabilityProvider::recruit( + provider.clone(), + Limits::default(), + 64 << 20, + 1024, + ) + .await + .unwrap(); + assert!( + recruited.is_none(), + "startup recovery must gate recruitment" + ); + assert!( + fixture + .directory + .load(session, now_ms()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + node.install_node_durability_provider( + provider, + NodeDurabilitySupervisorConfig::new( + beyonddb::APPLICATION_ID, + Limits::default(), + 64 << 20, + 1024, + Duration::from_millis(100), + Duration::from_secs(300), + 100_000, + ) + .unwrap(), + ) + .unwrap(); + ready.store(true, Ordering::Release); + } + node.start().unwrap(); + Self { + node, + metrics, + _tasks: tasks, + provisioner, + guard, + session, + id, + crash, + server, + } + } + + async fn shutdown(self) { + self.node.shutdown().await.unwrap(); + self.server.abort(); + let _ = self.server.await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn follower_fsync_acknowledges_signed_write_without_object_publication() { + assert_follower_durability(true).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn object_only_signed_write_waits_for_object_publication() { + assert_follower_durability(false).await; +} + +async fn assert_follower_durability(durability: bool) { + let withheld = Arc::new(WithheldPublication::new()); + let fixture = Fixture::with_store(1, withheld.clone()).await; + let second_follower = LogNode::new(&fixture, 200, true, false).await; + let follower = LogNode::new(&fixture, 201, true, false).await; + let leader = LogNode::new(&fixture, 202, false, durability).await; + if durability { + tokio::time::timeout(Duration::from_secs(5), async { + while leader.node.runtime().node_durability().is_none() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + } + let table = super::provisioning::table_id(&fixture, "Residency").await; + let target = beyonddb::data_target("123456789012", &table, &[0; 16]).unwrap(); + fixture.data[0].0.drain().await.unwrap(); + leader + .provisioner + .admit_existing_partition("123456789012", &table, &[0; 16]) + .await + .unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let before = authority.load(target.cell_id()).await.unwrap().unwrap(); + let predecessor = before.value().root.as_ref().unwrap().commit_sequence; + withheld.withhold(target.cell_id().as_bytes()); + let sdk = super::provisioning::sdk_without_retries(&fixture); + let metrics_before = leader.metrics.snapshot(); + let item = HashMap::from([ + ("id".into(), AwsAttributeValue::S("follower-only".into())), + ("value".into(), AwsAttributeValue::S("acknowledged".into())), + ]); + let result = tokio::time::timeout( + Duration::from_secs(5), + sdk.put_item() + .table_name("Residency") + .set_item(Some(item.clone())) + .send(), + ) + .await; + if !durability { + assert_eq!( + leader.metrics.snapshot()["command_responses"]["fleet"]["count"], + metrics_before["command_responses"]["fleet"]["count"], + "object-only timeout must not be counted as a follower acknowledgement" + ); + assert!( + result.is_err(), + "object-only write acknowledged before publication" + ); + assert!(withheld.blocked.is_cancelled()); + let current = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!( + current.value().root.as_ref().unwrap().commit_sequence, + predecessor + ); + withheld.release.cancel(); + leader.shutdown().await; + follower.shutdown().await; + second_follower.shutdown().await; + fixture.shutdown().await; + return; + } + if result.is_err() { + withheld.release.cancel(); + } + assert!( + result.is_ok(), + "signed write still waits for object publication despite an eligible follower" + ); + result.unwrap().unwrap(); + let metrics_after = leader.metrics.snapshot(); + assert!( + metrics_after["command_responses"]["fleet"]["count"] + .as_u64() + .unwrap() + > metrics_before["command_responses"]["fleet"]["count"] + .as_u64() + .unwrap(), + "the signed SDK acknowledgement must be counted as a follower response" + ); + tokio::time::timeout(Duration::from_secs(5), withheld.blocked.cancelled()) + .await + .unwrap(); + let after = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!( + after.value().root.as_ref().unwrap().commit_sequence, + predecessor, + "the acknowledged item must still be untiered" + ); + let enrolled = fixture + .directory + .load(leader.session, now_ms()) + .await + .unwrap() + .unwrap(); + assert!(enrolled.advertisement().log().unwrap().active()); + assert_eq!( + enrolled.advertisement().log().unwrap().members(), + &[second_follower.id, follower.id] + ); + + // Fence the lost owner while its immutable publication is withheld. This is + // a component loss test; a separate process-kill SDK test is still required. + leader.guard.fence(); + leader.crash.cancel(); + leader.server.abort(); + tokio::time::timeout(Duration::from_secs(20), async { + while fixture + .directory + .is_live(leader.session, now_ms()) + .await + .unwrap() + { + tokio::time::sleep(Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + let lost = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!( + lost.value().root.as_ref().unwrap().commit_sequence, + predecessor, + "owner loss must precede object-root publication" + ); + withheld.release.cancel(); + let successor = LogNode::new(&fixture, 203, false, false).await; + successor + .provisioner + .takeover_expired_partition("123456789012", &table, &[0; 16], &fixture.directory) + .await + .unwrap(); + let recovered = sdk + .get_item() + .table_name("Residency") + .key("id", item["id"].clone()) + .consistent_read(true) + .send() + .await + .unwrap(); + assert_eq!(recovered.item(), Some(&item)); + assert!( + fixture + .directory + .takeover_proof(leader.session, successor.session, now_ms()) + .await + .unwrap() + .is_some() + ); + successor.shutdown().await; + follower.shutdown().await; + second_follower.shutdown().await; + fixture.shutdown().await; +} diff --git a/tests/peer_network/residency/forward_cache.rs b/tests/peer_network/residency/forward_cache.rs new file mode 100644 index 0000000..5e27d57 --- /dev/null +++ b/tests/peer_network/residency/forward_cache.rs @@ -0,0 +1,547 @@ +use super::*; +use cellule_runtime::{ + codec::{BoundedEncoder, WireValue}, + identity::CellTarget, + peer::{PeerOperation, decode_peer_reply, wire}, +}; +use std::sync::atomic::{AtomicUsize, Ordering}; + +#[derive(Debug, Default)] +pub(super) struct CountedAuthority { + inner: Arc, + get_paths: std::sync::Mutex>, + failed_get_paths: std::sync::Mutex>, + pub(super) path: std::sync::Mutex>, + pub(super) reads: AtomicUsize, + pub(super) all_reads: AtomicUsize, + pub(super) creation_gate: std::sync::Mutex>, + pub(super) read_gate: std::sync::Mutex>, + pub(super) publication_gate: std::sync::Mutex>, +} + +#[derive(Clone, Debug)] +pub(super) struct CreationGate { + pub(super) paths: std::collections::HashSet, + pub(super) entered: Arc, + pub(super) release: Arc, +} + +impl std::fmt::Display for CountedAuthority { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("CountedAuthority") + } +} + +#[async_trait::async_trait] +impl object_store::ObjectStore for CountedAuthority { + async fn put_opts( + &self, + location: &object_store::path::Path, + payload: object_store::PutPayload, + options: object_store::PutOptions, + ) -> object_store::Result { + let gate = self + .creation_gate + .lock() + .unwrap() + .clone() + .filter(|gate| { + matches!(options.mode, object_store::PutMode::Create) + && gate.paths.contains(location) + }) + .or_else(|| { + self.publication_gate + .lock() + .unwrap() + .clone() + .filter(|gate| { + matches!(options.mode, object_store::PutMode::Update(_)) + && gate.paths.contains(location) + }) + }); + if let Some(gate) = gate { + gate.entered.add_permits(1); + gate.release.acquire().await.unwrap().forget(); + } + self.inner.put_opts(location, payload, options).await + } + + async fn put_multipart_opts( + &self, + location: &object_store::path::Path, + options: object_store::PutMultipartOptions, + ) -> object_store::Result> { + self.inner.put_multipart_opts(location, options).await + } + + async fn get_opts( + &self, + location: &object_store::path::Path, + options: object_store::GetOptions, + ) -> object_store::Result { + *self + .get_paths + .lock() + .unwrap() + .entry(location.clone()) + .or_default() += 1; + let gate = { + let mut gate = self.read_gate.lock().unwrap(); + if gate + .as_ref() + .is_some_and(|gate| gate.paths.contains(location)) + { + gate.take() + } else { + None + } + }; + if let Some(gate) = gate { + gate.entered.add_permits(1); + gate.release.acquire().await.unwrap().forget(); + } + if self.failed_get_paths.lock().unwrap().contains(location) { + return Err(object_store::Error::Generic { + store: "CountedAuthority", + source: std::io::Error::other("injected fresh enrollment read failure").into(), + }); + } + self.all_reads.fetch_add(1, Ordering::SeqCst); + let counted = self.path.lock().unwrap().as_ref() == Some(location); + if counted { + self.reads.fetch_add(1, Ordering::SeqCst); + tokio::time::sleep(std::time::Duration::from_millis(20)).await; + } + self.inner.get_opts(location, options).await + } + + fn delete_stream( + &self, + locations: futures_util::stream::BoxStream< + 'static, + object_store::Result, + >, + ) -> futures_util::stream::BoxStream<'static, object_store::Result> + { + self.inner.delete_stream(locations) + } + + fn list( + &self, + prefix: Option<&object_store::path::Path>, + ) -> futures_util::stream::BoxStream<'static, object_store::Result> + { + self.inner.list(prefix) + } + + async fn list_with_delimiter( + &self, + prefix: Option<&object_store::path::Path>, + ) -> object_store::Result { + self.inner.list_with_delimiter(prefix).await + } + + async fn copy_opts( + &self, + from: &object_store::path::Path, + to: &object_store::path::Path, + options: object_store::CopyOptions, + ) -> object_store::Result<()> { + self.inner.copy_opts(from, to, options).await + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn signed_forward_cache_skips_authority_io_and_rejects_drained_owner() { + for enabled in [false, true] { + let store = Arc::new(CountedAuthority::default()); + let fixture = + Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, enabled).await; + let remote = super::provisioning::Remote::new(&fixture).await; + let table = fixture + .client + .query::( + &account_target("123456789012").unwrap(), + None, + Json("Residency".into()), + ) + .await + .unwrap() + .output + .0 + .unwrap(); + let mut handle = fixture.data[0].0.clone(); + let entry = handle.catalog().entry(); + let target = CellTarget::new( + account_target("123456789012").unwrap().tenant(), + beyonddb::APPLICATION_ID, + entry.namespace(), + entry.partition(), + ) + .unwrap(); + let expected = wire::CellDescription { + cell_id: handle.cell_id().as_bytes().to_vec(), + incarnation: handle.incarnation().as_bytes().to_vec(), + code: handle.code().as_bytes().to_vec(), + schema: handle.schema(), + }; + let transport = PeerHttpRoundTrip::new( + Arc::new(BeyonddbPeerScope), + CellAuthority::new(fixture.layout.clone()), + fixture.directory.clone(), + Arc::new(fixture.remote_tls.client_identity()), + remote.session, + ); + let signer = PeerSigner::new( + remote.session, + fixture.application.registry().release_digest(), + fixture.remote_tls.signing_key().clone(), + ); + let principal = PeerPrincipal { + issuer: format!( + "beyonddb-peer:{}", + fixture + .directory + .fleet() + .as_bytes() + .iter() + .map(|b| format!("{b:02x}")) + .collect::() + ), + subject: remote + .session + .as_bytes() + .iter() + .map(|b| format!("{b:02x}")) + .collect(), + actions: vec!["beyonddb.cell.invoke".into()], + }; + let destination = fixture + .directory + .load(fixture.session, now_ms()) + .await + .unwrap() + .unwrap() + .advertisement() + .clone(); + let mut encoder = BoundedEncoder::new(64).unwrap(); + Json(()).encode(&mut encoder).unwrap(); + let input = encoder.finish(); + let send = |authorized: bool| { + let mut principal = principal.clone(); + if !authorized { + principal.actions = vec!["beyonddb.cell.invalid".into()]; + } + let now = now_ms(); + let request = signer + .sign( + principal, + now, + now + 60_000, + 30_000, + PeerOperation::Read(wire::ReadRequest { + target: Some(wire::Target { + tenant_id: target.tenant().as_bytes().to_vec(), + application_id: target.application().as_bytes().to_vec(), + namespace_id: target.namespace().as_bytes().to_vec(), + partition: target.partition().to_vec(), + }), + timeout_ms: 30_000, + minimum: None, + expected: Some(expected.clone()), + operation: Some(wire::read_request::Operation::CellQuery( + wire::CellQuery { + query_id: 4, + codec_version: 1, + input: input.clone(), + }, + )), + }), + ) + .unwrap(); + let transport = &transport; + let target = target.clone(); + let destination = destination.clone(); + async move { + let reply = transport + .send_to_node(target, destination, request, 30_000) + .await?; + Ok::<_, cellule_runtime::Error>(decode_peer_reply(&reply)?.outcome.unwrap()) + } + }; + *store.path.lock().unwrap() = + Some(fixture.layout.control_path(handle.cell_id().as_bytes())); + let warm = send(true).await.unwrap(); + assert!( + matches!(warm, wire::peer_reply::Outcome::Read(_)), + "warm read: {warm:?}" + ); + store.reads.store(0, Ordering::SeqCst); + let started = std::time::Instant::now(); + for _ in 0..4 { + assert!(matches!( + send(true).await.unwrap(), + wire::peer_reply::Outcome::Read(_) + )); + } + let reads = store.reads.load(Ordering::SeqCst); + println!( + "peer cache enabled={enabled}: four signed SQL reads in {:?}, authority reads={reads}", + started.elapsed() + ); + assert_eq!(reads, if enabled { 0 } else { 4 }); + assert!( + matches!(send(false).await.unwrap(), wire::peer_reply::Outcome::Error(error) + if error.code == wire::error::Code::PermissionDenied as i32) + ); + if enabled { + tokio::time::sleep(std::time::Duration::from_millis(550)).await; + store.reads.store(0, Ordering::SeqCst); + assert!(matches!( + send(true).await.unwrap(), + wire::peer_reply::Outcome::Read(_) + )); + assert_eq!( + store.reads.load(Ordering::SeqCst), + 0, + "expired cache should resolve the still-resident actor without authority I/O" + ); + } + if enabled { + fixture + .client + .query::(&target, None, Json(())) + .await + .unwrap(); + } + if enabled { + // Keep the receiver's cached handle while the same Cell closes and + // reopens. Matching code/schema/incarnation cannot validate an old + // owner epoch; the next signed invocation needs the new capability. + let before = handle.owner_fence(); + let started = std::time::Instant::now(); + handle.drain().await.unwrap(); + handle = fixture + .provisioner + .admit_existing_partition("123456789012", &table.id, &[0; 16]) + .await + .unwrap(); + assert_eq!(handle.owner_fence().incarnation, before.incarnation); + assert!(handle.owner_fence().epoch > before.epoch); + let reacquired = send(true).await; + println!( + "cached owner epoch reacquired in {:?}: succeeded={}", + started.elapsed(), + matches!(reacquired, Ok(wire::peer_reply::Outcome::Read(_))) + ); + assert!(matches!(reacquired, Ok(wire::peer_reply::Outcome::Read(_)))); + } + handle.drain().await.unwrap(); + let drained = send(true).await; + assert!( + matches!(drained, Err(cellule_runtime::Error::CellNotActive)), + "drained read: {drained:?}" + ); + let owner = CellAuthority::new(fixture.layout.clone()) + .load(handle.cell_id()) + .await + .unwrap() + .unwrap(); + assert!( + owner.value().owner.is_none(), + "forwarded invocation must not acquire a drained Cell" + ); + if enabled { + fixture + .client + .query::(&target, None, Json(())) + .await + .unwrap(); + let restored = CellAuthority::new(fixture.layout.clone()) + .load(handle.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(restored.value().owner.is_some()); + assert!(restored.value().root.is_some()); + assert_eq!( + restored.value().state, + cellule_runtime::control::ControlState::Serving + ); + } + *store.path.lock().unwrap() = None; + remote.shutdown().await; + fixture.shutdown().await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn remote_read_overlaps_fresh_authority_and_owner_enrollment() { + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let remote = super::provisioning::Remote::new(&fixture).await; + // Only the sender uses this observer; owner heartbeats and receiver + // authorization use the original store and cannot satisfy the probe. + let sender_store = Arc::new(CountedAuthority { + inner: store.inner.clone(), + ..CountedAuthority::default() + }); + let account = account_target("123456789012").unwrap(); + let layout = CellStorageLayout::new( + Store::new(sender_store.clone()), + object_store::path::Path::from("beyonddb-residency"), + *account.application().as_bytes(), + ); + let directory = NodeDirectory::new( + layout.clone(), + fixture.directory.fleet(), + Digest::from_bytes([90; 32]), + fixture.application.registry().release_digest(), + ); + let peers = BeyonddbPeers::new( + &remote.node, + layout.clone(), + directory, + remote.session, + &fixture.remote_tls, + ) + .unwrap(); + let client = peers.client_with_cache(remote.provisioner.clone(), true); + let handle = &fixture.data[0].0; + let entry = handle.catalog().entry(); + let target = CellTarget::new( + account.tenant(), + beyonddb::APPLICATION_ID, + entry.namespace(), + entry.partition(), + ) + .unwrap(); + let expected = client + .query::(&target, None, Json(())) + .await + .unwrap() + .output; + let owner_path = layout.node_path(fixture.session.as_bytes()); + let before = sender_store + .get_paths + .lock() + .unwrap() + .get(&owner_path) + .copied() + .unwrap_or_default(); + let gate = CreationGate { + paths: std::collections::HashSet::from([layout.control_path(handle.cell_id().as_bytes())]), + entered: Arc::new(tokio::sync::Semaphore::new(0)), + release: Arc::new(tokio::sync::Semaphore::new(0)), + }; + *sender_store.read_gate.lock().unwrap() = Some(gate.clone()); + let read = tokio::spawn({ + let client = client.clone(); + let target = target.clone(); + async move { + client + .query::(&target, None, Json(())) + .await + } + }); + tokio::time::timeout(std::time::Duration::from_secs(2), gate.entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let overlap = tokio::time::timeout(std::time::Duration::from_millis(500), async { + loop { + if sender_store + .get_paths + .lock() + .unwrap() + .get(&owner_path) + .copied() + .unwrap_or_default() + > before + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .is_ok(); + gate.release.add_permits(1); + let actual = read.await.unwrap().unwrap().output; + assert_eq!(actual, expected); + let after = sender_store + .get_paths + .lock() + .unwrap() + .get(&owner_path) + .copied() + .unwrap_or_default(); + sender_store + .failed_get_paths + .lock() + .unwrap() + .insert(owner_path.clone()); + let failed = client + .query::(&target, None, Json(())) + .await; + assert!( + failed.is_err(), + "a current owner's fresh probe must fail closed" + ); + let replacement = + super::provisioning::Remote::with_identity(&fixture, SessionId::from_bytes([97; 16]), 100) + .await; + + // Keep the failed old-session probe, but change authority while its read + // is gated. The new remote owner must not inherit that obsolete failure. + *sender_store.read_gate.lock().unwrap() = Some(gate.clone()); + let moved_read = tokio::spawn({ + let client = client.clone(); + let target = target.clone(); + async move { + client + .query::(&target, None, Json(())) + .await + } + }); + tokio::time::timeout(std::time::Duration::from_secs(2), gate.entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let spec = &expected.0.as_ref().unwrap().spec; + handle.drain().await.unwrap(); + replacement + .provisioner + .admit_existing_partition("123456789012", &spec.table.id, &spec.partition_id) + .await + .unwrap(); + gate.release.add_permits(1); + assert_eq!(moved_read.await.unwrap().unwrap().output, expected); + let owner = CellAuthority::new(fixture.layout.clone()) + .load(handle.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!( + owner.value().owner.as_ref().unwrap().session, + replacement.session + ); + let item = super::provisioning::sdk_without_retries(&fixture) + .get_item() + .table_name("Residency") + .key("id", fixture.data[0].1["id"].clone()) + .consistent_read(true) + .send() + .await + .unwrap(); + assert_eq!(item.item.as_ref(), Some(&fixture.data[0].1)); + replacement.shutdown().await; + remote.shutdown().await; + fixture.shutdown().await; + println!( + "fresh owner enrollment entered during gated authority read={overlap}; sender owner GETs {before}->{after}" + ); + assert!(overlap, "fresh owner enrollment waited for authority I/O"); +} diff --git a/tests/peer_network/residency/live_owner_failover.rs b/tests/peer_network/residency/live_owner_failover.rs new file mode 100644 index 0000000..2a89e2d --- /dev/null +++ b/tests/peer_network/residency/live_owner_failover.rs @@ -0,0 +1,392 @@ +use super::*; +use beyonddb::{ + BeginCrossCellTransaction, BeginCrossCellTransactionInput, CellStorage, CoordinatorDecision, + CoordinatorParticipant, CoordinatorParticipantTarget, CoordinatorPhaseInput, + CoordinatorProvisioner, DecideCrossCellTransaction, DecideCrossCellTransactionInput, + IndexedTransactionOperation, PreparePartitionTransaction, PreparePartitionTransactionInput, + PutItemInput, ReadCrossCellTransaction, ReadCrossCellTransactionInput, + RecordParticipantPrepare, ResolvePartitionTransaction, ResolveTransactionInput, + TransactionOperation, +}; +use cellule_runtime::identity::RequestId; +use extenddb_core::types::{AttributeValue, Item}; + +fn identity() -> cellule_runtime::MutationIdentity { + let now = now_ms(); + cellule_runtime::MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + issued_at_ms: now, + expires_at_ms: now + 60_000, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn serving_recovery_preserves_live_coordinator_then_finishes_after_owner_expiry() { + let fixture = Fixture::new().await; + let remote = provisioning::Remote::new(&fixture).await; + let account = account_target("123456789012").unwrap(); + let table = fixture + .client + .query::(&account, None, Json("Residency".into())) + .await + .unwrap() + .output + .0 + .unwrap(); + let route = crate::single_leaf_route(&fixture.client, &account, &table.id) + .await + .unwrap(); + let mut participants = Vec::new(); + for (partition, (original, item)) in route.partitions.iter().zip(&fixture.data) { + let target = + beyonddb::data_target("123456789012", &table.id, &partition.partition_id).unwrap(); + assert_eq!(target.cell_id(), original.cell_id()); + original.drain().await.unwrap(); + remote + .provisioner + .admit_existing_partition("123456789012", &table.id, &partition.partition_id) + .await + .unwrap(); + let id = item["id"].as_s().unwrap().clone(); + participants.push(( + target, + CoordinatorParticipant { + target: CoordinatorParticipantTarget::Data { + table_id: table.id.clone(), + partition_id: partition.partition_id, + epoch: partition.epoch, + }, + operations: vec![IndexedTransactionOperation { + index: u8::try_from(participants.len()).unwrap(), + operation: TransactionOperation::Put(PutItemInput { + table_name: "Residency".into(), + table_id: table.id.clone(), + item: Item::from([ + ("id".into(), AttributeValue::S(id)), + ( + "value".into(), + AttributeValue::S("owner-expiry-recovered".into()), + ), + ]), + condition: None, + }), + }], + }, + )); + } + assert_eq!(participants.len(), 2); + participants.sort_by_key(|(target, _)| *target.cell_id().as_bytes()); + fixture + .provisioner + .install_transaction_recovery_loop( + &fixture.tasks, + CellStorage::new(fixture.client.clone(), "us-east-1"), + fixture.directory.clone(), + vec!["123456789012".into()], + ) + .unwrap(); + let transaction_id = *uuid::Uuid::now_v7().as_bytes(); + let coordinator = beyonddb::coordinator_target("123456789012", &transaction_id).unwrap(); + let client = remote.client(&fixture); + remote + .provisioner + .ensure(&client, "123456789012", &transaction_id) + .await + .unwrap(); + client + .command::( + &coordinator, + identity(), + Json(beyonddb::TransactionCommandInput::Inline( + BeginCrossCellTransactionInput { + account_id: "123456789012".into(), + transaction_id, + token: None, + participants: participants + .iter() + .map(|(_, participant)| participant.clone()) + .collect(), + }, + )), + ) + .await + .unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let before = authority + .load(coordinator.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + assert_eq!(before.owner.as_ref().unwrap().session, remote.session); + for (position, (target, participant)) in participants.iter().enumerate() { + let CoordinatorParticipantTarget::Data { + table_id, epoch, .. + } = &participant.target + else { + unreachable!() + }; + let prepared = client + .command::( + target, + identity(), + Json(beyonddb::TransactionCommandInput::Inline( + PreparePartitionTransactionInput { + table_id: table_id.clone(), + epoch: *epoch, + transaction_id, + coordinator_cell: *coordinator.cell_id().as_bytes(), + coordinator_key: transaction_id.to_vec(), + operations: participant + .operations + .iter() + .map(|operation| operation.operation.clone()) + .collect(), + }, + )), + ) + .await + .unwrap(); + client + .command::( + &coordinator, + identity(), + Json(CoordinatorPhaseInput { + account_id: "123456789012".into(), + transaction_id, + routing_key: transaction_id.to_vec(), + position: u8::try_from(position).unwrap(), + participant_cell: *target.cell_id().as_bytes(), + sequence: prepared.receipt.commit_sequence, + }), + ) + .await + .unwrap(); + } + client + .command::( + &coordinator, + identity(), + Json(DecideCrossCellTransactionInput { + account_id: "123456789012".into(), + transaction_id, + routing_key: transaction_id.to_vec(), + decision: CoordinatorDecision::Commit, + }), + ) + .await + .unwrap(); + client + .command::( + &participants[0].0, + identity(), + Json(ResolveTransactionInput { + transaction_id, + coordinator_cell: *coordinator.cell_id().as_bytes(), + commit: true, + }), + ) + .await + .unwrap(); + // No participant completion receipt is recorded. Recovery was installed + // before discovery registration, but must leave this live owner alone. + tokio::time::sleep(std::time::Duration::from_secs(2)).await; + let live = authority + .load(coordinator.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + assert!( + fixture + .directory + .is_live(remote.session, now_ms()) + .await + .unwrap() + ); + assert_eq!(live.incarnation, before.incarnation); + assert_eq!(live.epoch, before.epoch); + assert_eq!(live.owner.as_ref().unwrap().session, remote.session); + let input = ReadCrossCellTransactionInput { + account_id: "123456789012".into(), + transaction_id, + routing_key: transaction_id.to_vec(), + }; + let pending = client + .query::(&coordinator, None, Json(input.clone())) + .await + .unwrap() + .output + .0 + .unwrap(); + assert_eq!(pending.decision, CoordinatorDecision::Commit); + assert_eq!(pending.resolved_count, 0); + remote.stop_listener(); + remote.lease.cancel(); + tokio::time::timeout(std::time::Duration::from_secs(30), async { + while fixture + .directory + .is_live(remote.session, now_ms()) + .await + .unwrap() + { + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + // Private status reads cannot resolve participants. Both completion receipts + // must be recorded by the already-serving recovery worker before SDK reads. + tokio::time::timeout(std::time::Duration::from_secs(45), async { + loop { + if let Ok(status) = fixture + .client + .query::(&coordinator, None, Json(input.clone())) + .await + { + let status = status.output.0.unwrap(); + assert_eq!(status.decision, CoordinatorDecision::Commit); + if status.resolved_count == 2 { + break; + } + } + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .await + .expect("serving recovery must discover the expired owner and record both resolutions"); + for (_, item) in &fixture.data { + let result = provisioning::sdk_without_retries(&fixture) + .get_item() + .table_name("Residency") + .key("id", item["id"].clone()) + .consistent_read(true) + .send() + .await + .unwrap(); + assert_eq!( + result.item.unwrap()["value"], + AwsAttributeValue::S("owner-expiry-recovered".into()) + ); + } + let recovered = authority + .load(coordinator.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + assert_eq!(recovered.incarnation, before.incarnation); + assert!(recovered.epoch > before.epoch); + assert_eq!(recovered.owner.as_ref().unwrap().session, fixture.session); + assert!(matches!( + remote.node.shutdown().await, + Ok(()) | Err(cellule_runtime::Error::Fenced) + )); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn published_idle_range_can_move_while_its_former_node_stays_live() { + let fixture = Fixture::with_partition_count(1).await; + let remote = provisioning::Remote::new(&fixture).await; + let account = account_target("123456789012").unwrap(); + let table = fixture + .client + .query::(&account, None, Json("Residency".into())) + .await + .unwrap() + .output + .0 + .unwrap(); + let route = crate::single_leaf_route(&fixture.client, &account, &table.id) + .await + .unwrap(); + let partition = &route.partitions[0]; + let target = beyonddb::data_target("123456789012", &table.id, &partition.partition_id).unwrap(); + fixture.data[0].0.drain().await.unwrap(); + let handle = remote + .provisioner + .admit_existing_partition("123456789012", &table.id, &partition.partition_id) + .await + .unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let before = authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + fixture + .provisioner + .recover_registered_partitions("123456789012", &fixture.client, &fixture.directory) + .await + .unwrap(); + let live = authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + assert_eq!(live.owner.as_ref().unwrap().session, remote.session); + assert_eq!(live.epoch, before.epoch); + handle.drain().await.unwrap(); + let idle = authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + assert!( + fixture + .directory + .is_live(remote.session, now_ms()) + .await + .unwrap() + ); + assert_eq!(idle.state, cellule_runtime::control::ControlState::Idle); + assert!(idle.owner.is_none()); + fixture + .provisioner + .recover_registered_partitions("123456789012", &fixture.client, &fixture.directory) + .await + .unwrap(); + let restored = authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + let expected = fixture.data[0].1.clone(); + let read = provisioning::sdk_without_retries(&fixture) + .get_item() + .table_name("Residency") + .key("id", expected["id"].clone()) + .consistent_read(true) + .send() + .await; + let source_still_live = fixture + .directory + .is_live(remote.session, now_ms()) + .await + .unwrap(); + let destination = fixture.session; + remote.shutdown().await; + fixture.shutdown().await; + assert!(source_still_live); + crate::peer_network::recovery::assert_retained_range( + &before, + &restored, + destination, + live.owner.as_ref().unwrap().session, + ); + assert_eq!(restored.owner.as_ref().unwrap().session, destination); + assert!(restored.epoch > before.epoch); + assert_eq!(read.unwrap().item, Some(expected)); +} diff --git a/tests/peer_network/residency/prepare_batch.rs b/tests/peer_network/residency/prepare_batch.rs new file mode 100644 index 0000000..b48796a --- /dev/null +++ b/tests/peer_network/residency/prepare_batch.rs @@ -0,0 +1,199 @@ +use super::*; +use beyonddb::{ + ParticipantTransactionState, PreparePartitionBatchOutcome, PreparePartitionTransactionBatch, + PreparePartitionTransactionInput, PutItemInput, ReadPartitionTransaction, ReadTransactionInput, + ResolvePartitionTransaction, ResolveTransactionInput, ResolveTransactionOutcome, + TransactionOperation, +}; +use cellule_runtime::{MutationIdentity, client::InvocationError, identity::RequestId}; + +fn identity() -> MutationIdentity { + let issued_at_ms = now_ms(); + MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + issued_at_ms, + expires_at_ms: issued_at_ms + 60_000, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn prepare_batch_rolls_back_locks_and_restores_independent_intents() { + let fixture = Fixture::with_capacity(1, 16).await; + let remote = super::provisioning::Remote::new(&fixture).await; + let client = remote.client(&fixture); + let account = account_target("123456789012").unwrap(); + let table_id = super::provisioning::table_id(&fixture, "Residency").await; + let range = super::cell_models::ranges(&fixture, "Residency") + .await + .remove(0); + let target = beyonddb::data_target("123456789012", &table_id, &range.partition_id).unwrap(); + let coordinator_cell = *account.cell_id().as_bytes(); + let input = |index: u8, key: &str| PreparePartitionTransactionInput { + table_id: table_id.clone(), + epoch: range.epoch, + transaction_id: [index; 16], + coordinator_cell, + coordinator_key: vec![index; 16], + operations: vec![TransactionOperation::Put(PutItemInput { + table_name: "Residency".into(), + table_id: table_id.clone(), + item: Item::from([("id".into(), AttributeValue::S(key.into()))]), + condition: None, + })], + }; + let first = input(180, "batched-first"); + let second = input(181, "batched-second"); + let read = |input: &PreparePartitionTransactionInput| { + Json(ReadTransactionInput { + transaction_id: input.transaction_id, + coordinator_cell, + }) + }; + // The second prepare conflicts with the first staged lock. The rejection + // must roll back the first intent and lock, not just the conflicting child. + let mut conflicting = second.clone(); + conflicting.operations = first.operations.clone(); + let rejected = client + .command::( + &target, + identity(), + Json(vec![first.clone(), conflicting]), + ) + .await; + assert!(matches!(rejected, Err(InvocationError::Rejected(result)) + if result.output.0 == PreparePartitionBatchOutcome::IndividualRequired)); + for input in [&first, &second] { + assert_eq!( + client + .query::(&target, None, read(input)) + .await + .unwrap() + .output + .0, + ParticipantTransactionState::Missing + ); + } + // Identical transaction identities within a batch are rejected before any + // prepare; they must not accidentally share one coordinator receipt. + let duplicate = client + .command::( + &target, + identity(), + Json(vec![first.clone(), first.clone()]), + ) + .await; + assert!(matches!(duplicate, Err(InvocationError::Rejected(result)) + if result.output.0 == PreparePartitionBatchOutcome::IndividualRequired)); + let mutation = identity(); + let inputs = Json(vec![first.clone(), second.clone()]); + let prepared = client + .command::(&target, mutation, inputs.clone()) + .await + .unwrap(); + assert_eq!(prepared.output.0, PreparePartitionBatchOutcome::Prepared); + let replay = client + .command::(&target, mutation, inputs.clone()) + .await + .unwrap(); + assert_eq!(replay.receipt, prepared.receipt); + // A *new* command seeing existing intents requests individual handling; + // the original durable intents survive that rejected application savepoint. + assert!( + matches!(client.command::(&target, identity(), inputs).await, + Err(InvocationError::Rejected(result)) if result.output.0 == PreparePartitionBatchOutcome::IndividualRequired) + ); + fixture.data[0].0.drain().await.unwrap(); + remote + .provisioner + .admit_existing_partition("123456789012", &table_id, &range.partition_id) + .await + .unwrap(); + for input in [&first, &second] { + assert_eq!( + client + .query::(&target, None, read(input)) + .await + .unwrap() + .output + .0, + ParticipantTransactionState::Prepared + ); + } + // Shared publication never merges terminal decisions. Commit one intent + // and abort the other, then restore again and verify both durable states. + for (input, commit) in [(&first, true), (&second, false)] { + let result = client + .command::( + &target, + identity(), + Json(ResolveTransactionInput { + transaction_id: input.transaction_id, + coordinator_cell, + commit, + }), + ) + .await + .unwrap(); + assert_eq!( + result.output.0, + if commit { + ResolveTransactionOutcome::Committed + } else { + ResolveTransactionOutcome::Aborted + } + ); + } + remote + .provisioner + .admit_existing_partition("123456789012", &table_id, &range.partition_id) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_existing_partition("123456789012", &table_id, &range.partition_id) + .await + .unwrap(); + for (input, expected) in [ + (&first, ParticipantTransactionState::Committed), + (&second, ParticipantTransactionState::Aborted), + ] { + assert_eq!( + fixture + .client + .query::(&target, None, read(input)) + .await + .unwrap() + .output + .0, + expected + ); + } + for (key, exists) in [("batched-first", true), ("batched-second", false)] { + assert_eq!( + fixture + .sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S(key.into())) + .send() + .await + .unwrap() + .item + .is_some(), + exists + ); + fixture + .sdk + .put_item() + .table_name("Residency") + .item("id", AwsAttributeValue::S(key.into())) + .send() + .await + .unwrap(); + } + remote.shutdown().await; + fixture.shutdown().await; +} diff --git a/tests/peer_network/residency/prepare_capacity.rs b/tests/peer_network/residency/prepare_capacity.rs new file mode 100644 index 0000000..95abcbc --- /dev/null +++ b/tests/peer_network/residency/prepare_capacity.rs @@ -0,0 +1,365 @@ +use super::*; +use beyonddb::{ + ParticipantTransactionState, PreparePartitionTransaction, PreparePartitionTransactionBounded, + PreparePartitionTransactionInput, PrepareTransactionOutcome, PutItemInput, ReadPartitionState, + ReadPartitionTransaction, ReadTransactionInput, ResolvePartitionTransaction, + ResolveTransactionInput, TransactionCommandInput, TransactionOperation, +}; +use cellule_runtime::{MutationIdentity, identity::RequestId}; + +const ACCOUNT: &str = "123456789012"; + +fn identity() -> MutationIdentity { + let issued_at_ms = now_ms(); + MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + issued_at_ms, + expires_at_ms: issued_at_ms + 60_000, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_bounded_prepare_preserves_large_failure_images_and_token_recovery() { + use aws_sdk_dynamodb::types::{Put, ReturnValuesOnConditionCheckFailure, TransactWriteItem}; + + let fixture = Fixture::with_capacity(1, 16).await; + let sdk = super::provisioning::sdk_without_retries(&fixture); + let old = SdkItem::from([ + ("id".into(), AwsAttributeValue::S("wide-condition".into())), + ( + "payload".into(), + AwsAttributeValue::S("\0".repeat(384 * 1024)), + ), + ]); + sdk.put_item() + .table_name("Residency") + .set_item(Some(old.clone())) + .send() + .await + .unwrap(); + let write = sdk + .transact_write_items() + .client_request_token("bounded-wide-condition") + .transact_items( + TransactWriteItem::builder() + .put( + Put::builder() + .table_name("Residency") + .item("id", AwsAttributeValue::S("wide-staged".into())) + .item("value", AwsAttributeValue::S("committed".into())) + .build() + .unwrap(), + ) + .build(), + ) + .transact_items( + TransactWriteItem::builder() + .put( + Put::builder() + .table_name("Residency") + .item("id", old["id"].clone()) + .condition_expression("attribute_not_exists(id)") + .return_values_on_condition_check_failure( + ReturnValuesOnConditionCheckFailure::AllOld, + ) + .build() + .unwrap(), + ) + .build(), + ); + let error = write.clone().send().await.unwrap_err(); + let aws_sdk_dynamodb::operation::transact_write_items::TransactWriteItemsError::TransactionCanceledException(error) = error.as_service_error().unwrap() else { + panic!("expected condition cancellation"); + }; + assert_eq!( + error.cancellation_reasons()[1].code(), + Some("ConditionalCheckFailed") + ); + assert_eq!(error.cancellation_reasons()[1].item(), Some(&old)); + assert!( + sdk.get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("wide-staged".into())) + .consistent_read(true) + .send() + .await + .unwrap() + .item + .is_none() + ); + // Release and restore the data owner after the failed transaction. The old + // image remains intact and cancellation leaves no blocking write locks. + let id = super::provisioning::table_id(&fixture, "Residency").await; + let range = super::cell_models::ranges(&fixture, "Residency") + .await + .remove(0); + fixture.data[0].0.drain().await.unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + assert_eq!( + sdk.get_item() + .table_name("Residency") + .key("id", old["id"].clone()) + .consistent_read(true) + .send() + .await + .unwrap() + .item, + Some(old.clone()) + ); + sdk.delete_item() + .table_name("Residency") + .key("id", old["id"].clone()) + .send() + .await + .unwrap(); + // Canceled token reuse keeps the identical request fingerprint. + write.clone().send().await.unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + let item = sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("wide-staged".into())) + .consistent_read(true) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::S("committed".into())); + sdk.put_item() + .table_name("Residency") + .item("id", AwsAttributeValue::S("wide-staged".into())) + .item("value", AwsAttributeValue::S("newer".into())) + .send() + .await + .unwrap(); + write.send().await.unwrap(); + assert_eq!( + sdk.get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("wide-staged".into())) + .consistent_read(true) + .send() + .await + .unwrap() + .item + .unwrap()["value"], + AwsAttributeValue::S("newer".into()) + ); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn eight_in_flight_prepares_fit_the_participant_mailbox() { + // Preserve the old wide path as a control: its 4 MiB reply envelope admits + // only three held requests. The bounded path must admit all eight. + for bounded in [false, true] { + held_prepares(bounded).await; + } +} + +async fn held_prepares(bounded: bool) { + use super::forward_cache::{CountedAuthority, CreationGate}; + + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let remote = super::provisioning::Remote::new(&fixture).await; + let client = remote.client_with_cache(&fixture, true); + let handle = &fixture.data[0].0; + let entry = handle.catalog().entry(); + let account = account_target(ACCOUNT).unwrap(); + let target = cellule_runtime::identity::CellTarget::new( + account.tenant(), + beyonddb::APPLICATION_ID, + entry.namespace(), + entry.partition(), + ) + .unwrap(); + let spec = client + .query::(&target, None, Json(())) + .await + .unwrap() + .output + .0 + .unwrap() + .spec; + let coordinator_cell = *account.cell_id().as_bytes(); + let gate = CreationGate { + paths: std::collections::HashSet::from([fixture + .layout + .control_path(handle.cell_id().as_bytes())]), + entered: Arc::new(tokio::sync::Semaphore::new(0)), + release: Arc::new(tokio::sync::Semaphore::new(0)), + }; + *store.publication_gate.lock().unwrap() = Some(gate.clone()); + let before_retained = fixture.node.runtime().stats().retained_bytes(); + let mut tasks = Vec::new(); + for index in 0..8_u8 { + let client = client.clone(); + let target = target.clone(); + let transaction_id = [170 + index; 16]; + let input = PreparePartitionTransactionInput { + table_id: spec.table.id.clone(), + epoch: spec.epoch, + transaction_id, + coordinator_cell, + coordinator_key: transaction_id.to_vec(), + operations: vec![TransactionOperation::Put(PutItemInput { + table_name: "Residency".into(), + table_id: spec.table.id.clone(), + item: Item::from([ + ( + "id".into(), + AttributeValue::S(format!("prepare-pressure-{index}")), + ), + ("value".into(), AttributeValue::S("staged".into())), + ]), + condition: None, + })], + }; + tasks.push(tokio::spawn(async move { + ( + transaction_id, + if bounded { + client + .command::( + &target, + identity(), + Json(input), + ) + .await + } else { + client + .command::( + &target, + identity(), + Json(TransactionCommandInput::Inline(input)), + ) + .await + }, + ) + })); + if index == 0 { + // Hold the first real command's publication, and thus its mailbox + // reservation, while the remaining requests reach the owner. + tokio::time::timeout(std::time::Duration::from_secs(5), gate.entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + } + } + // Prove simultaneous admission before releasing publication. The node + // ledger includes each actual request's reserved reply bytes; a small + // bounded prepare batch cannot reach this threshold with fewer than eight. + let all_held = tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + let ready = if bounded { + fixture.node.runtime().stats().retained_bytes() >= before_retained + 8 * 64 * 1024 + } else { + tasks.iter().filter(|task| task.is_finished()).count() >= 5 + }; + if ready { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .is_ok(); + let held_retained = fixture.node.runtime().stats().retained_bytes(); + *store.publication_gate.lock().unwrap() = None; + gate.release.add_permits(64); + let mut accepted = Vec::new(); + let mut refused = Vec::new(); + for task in tasks { + let (transaction_id, result) = + tokio::time::timeout(std::time::Duration::from_secs(30), task) + .await + .unwrap() + .unwrap(); + match result { + Ok(result) => { + assert_eq!(result.output.0, PrepareTransactionOutcome::Prepared); + accepted.push(transaction_id); + } + Err(error) => { + assert!( + matches!( + error, + cellule_runtime::client::InvocationError::NotStarted( + cellule_runtime::Error::Capacity(_) + ) + ), + "unexpected prepare failure: {error:?}" + ); + refused.push(format!("{error:?}")); + } + } + } + // Check durable prepares after owner transfer, then abort every accepted + // intent before asserting the concurrency result or shutting down. + handle.drain().await.unwrap(); + remote + .provisioner + .admit_existing_partition(ACCOUNT, &spec.table.id, &spec.partition_id) + .await + .unwrap(); + for transaction_id in &accepted { + let observed = client + .query::( + &target, + None, + Json(ReadTransactionInput { + transaction_id: *transaction_id, + coordinator_cell, + }), + ) + .await + .unwrap(); + assert_eq!(observed.output.0, ParticipantTransactionState::Prepared); + client + .command::( + &target, + identity(), + Json(ResolveTransactionInput { + transaction_id: *transaction_id, + coordinator_cell, + commit: false, + }), + ) + .await + .unwrap(); + } + remote.shutdown().await; + fixture.shutdown().await; + println!( + "eight held prepares bounded={bounded}: accepted={}, held={all_held}, retained={before_retained}->{held_retained}, refused={refused:?}", + accepted.len() + ); + assert!( + all_held, + "requests did not reach admission while publication was held" + ); + assert_eq!( + accepted.len(), + if bounded { 8 } else { 3 }, + "prepare reply reservations refused: {refused:?}" + ); +} diff --git a/tests/peer_network/residency/pressure.rs b/tests/peer_network/residency/pressure.rs new file mode 100644 index 0000000..a9d4df2 --- /dev/null +++ b/tests/peer_network/residency/pressure.rs @@ -0,0 +1,99 @@ +use super::*; +use std::sync::atomic::{AtomicBool, Ordering}; + +#[derive(Default)] +struct RefusalGate { + armed: AtomicBool, + refused: tokio::sync::Notify, + release: tokio::sync::Notify, +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn credential_query_waits_for_peer_memory_and_reads_revocation_after_release() { + const ACCESS_KEY: &str = "AKIAIOSFODNN7EXAMPLE"; + let gate = Arc::new(RefusalGate::default()); + let intercepted = gate.clone(); + let fixture = Fixture::with_store_capacity_cache_and_router( + 1, + Arc::new(InMemory::new()), + 8, + false, + move |router| { + router.layer(axum::middleware::from_fn( + move |request: axum::extract::Request, next: axum::middleware::Next| { + let gate = intercepted.clone(); + async move { + let reply = next.run(request).await; + if reply.status() == axum::http::StatusCode::SERVICE_UNAVAILABLE + && gate.armed.swap(false, Ordering::SeqCst) + { + // Hold the actual pre-dispatch memory refusal until + // authority proves this owner has released the Cell. + gate.refused.notify_one(); + tokio::time::timeout( + std::time::Duration::from_secs(10), + gate.release.notified(), + ) + .await + .unwrap(); + } + reply + } + }, + )) + }, + ) + .await; + let remote = provisioning::Remote::new(&fixture).await; + let credentials = + CellCredentialStore::new(remote.client(&fixture), fixture.layout.clone(), [38; 32]); + let credential = fixture + .provisioner + .admit_credential(ACCESS_KEY) + .await + .unwrap(); + let runtime = fixture.node.runtime(); + let stats = runtime.stats(); + let memory = runtime + .try_reserve_node_bytes(stats.retained_capacity_bytes() - stats.retained_bytes()) + .unwrap(); + gate.armed.store(true, Ordering::SeqCst); + let lookup = credentials.lookup_credential(ACCESS_KEY); + tokio::pin!(lookup); + tokio::select! { + result = &mut lookup => panic!("lookup completed before the peer refusal: {}", result.is_ok()), + () = gate.refused.notified() => {} + } + drop(memory); + credential.drain().await.unwrap(); + let target = beyonddb::credential_target(ACCESS_KEY).unwrap(); + assert!( + CellAuthority::new(fixture.layout.clone()) + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .owner + .is_none() + ); + gate.release.notify_one(); + let record = tokio::time::timeout(std::time::Duration::from_secs(10), lookup) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(record.account_id, "123456789012"); + assert!(record.is_active); + assert!(credentials.revoke_credential(ACCESS_KEY).await.unwrap()); + assert!( + !credentials + .lookup_credential(ACCESS_KEY) + .await + .unwrap() + .unwrap() + .is_active + ); + remote.shutdown().await; + fixture.shutdown().await; +} diff --git a/tests/peer_network/residency/provisioning.rs b/tests/peer_network/residency/provisioning.rs index 35d8f82..52f78ca 100644 --- a/tests/peer_network/residency/provisioning.rs +++ b/tests/peer_network/residency/provisioning.rs @@ -12,14 +12,30 @@ use cellule_runtime::{ pub(super) struct Remote { pub(super) node: CellNode, _tasks: Arc, - provisioner: Arc, + pub(super) provisioner: Arc, pub(super) session: SessionId, endpoint: String, - lease: CancellationToken, + pub(super) lease: CancellationToken, server: tokio::task::JoinHandle<()>, } impl Remote { + pub(super) fn client(&self, fixture: &Fixture) -> CellClient { + self.client_with_cache(fixture, false) + } + + pub(super) fn client_with_cache(&self, fixture: &Fixture, enabled: bool) -> CellClient { + BeyonddbPeers::new( + &self.node, + fixture.layout.clone(), + fixture.directory.clone(), + self.session, + &fixture.remote_tls, + ) + .unwrap() + .client_with_cache(self.provisioner.clone(), enabled) + } + pub(super) async fn new(fixture: &Fixture) -> Self { Self::with_router(fixture, std::convert::identity).await } @@ -27,10 +43,38 @@ impl Remote { pub(super) async fn with_router( fixture: &Fixture, wrap: impl FnOnce(axum::Router) -> axum::Router, + ) -> Self { + Self::with_router_identity(fixture, wrap, SessionId::from_bytes([96; 16]), 98, false).await + } + + pub(super) async fn with_identity( + fixture: &Fixture, + session: SessionId, + node_byte: u8, + ) -> Self { + Self::with_router_identity(fixture, std::convert::identity, session, node_byte, false).await + } + + pub(super) async fn with_cache(fixture: &Fixture) -> Self { + Self::with_router_identity( + fixture, + std::convert::identity, + SessionId::from_bytes([96; 16]), + 98, + true, + ) + .await + } + + async fn with_router_identity( + fixture: &Fixture, + wrap: impl FnOnce(axum::Router) -> axum::Router, + session: SessionId, + node_byte: u8, + peer_cache: bool, ) -> Self { let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); let endpoint = format!("https://{}", listener.local_addr().unwrap()); - let session = SessionId::from_bytes([96; 16]); let lease = CancellationToken::new(); let (remote, _tasks) = start_node( fixture.application.clone(), @@ -39,7 +83,7 @@ impl Remote { session, endpoint.clone(), &fixture.remote_tls, - 98, + node_byte, lease.clone(), ) .await; @@ -60,14 +104,18 @@ impl Remote { fixture.layout.clone(), session, endpoint.clone(), - fixture._files.path().join("remote-data"), + fixture._files.path().join(if node_byte == 98 { + "remote-data".into() + } else { + format!("remote-data-{node_byte}") + }), ) .unwrap() .with_initial_partition_count(2) .unwrap() .with_peers(peers.clone()), ); - let router = wrap(peers.router(provisioner.clone())); + let router = wrap(peers.router_with_cache(provisioner.clone(), peer_cache)); let tls = LoadedPeerTls::load( &fixture._files.path().join("remote.crt"), &fixture._files.path().join("remote.key"), @@ -178,8 +226,151 @@ pub(super) fn sdk_without_retries(fixture: &Fixture) -> aws_sdk_dynamodb::Client } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn sdk_read_recovers_expired_directory_and_data_owners() { +async fn large_transaction_read_from_remote_owner_keeps_read_only_snapshot() { + use aws_sdk_dynamodb::types::{Get, TransactGetItem}; + let fixture = Fixture::with_partition_count(1).await; + let sdk = sdk_without_retries(&fixture); + let mut reads = Vec::new(); + let mut expected = Vec::new(); + // The DynamoDB aggregate is below 4 MiB, but base64 exceeds the Cell wire limit. + for index in 0..10 { + let key = std::collections::HashMap::from([( + "id".into(), + AwsAttributeValue::S(format!("large-{index}")), + )]); + let mut item = key.clone(); + item.insert( + "payload".into(), + AwsAttributeValue::B(vec![0xa5; 380 * 1024].into()), + ); + sdk.put_item() + .table_name("Residency") + .set_item(Some(item.clone())) + .send() + .await + .unwrap(); + reads.push( + TransactGetItem::builder() + .get( + Get::builder() + .table_name("Residency") + .set_key(Some(key)) + .build() + .unwrap(), + ) + .build(), + ); + expected.push(item); + } + let remote = Remote::new(&fixture).await; + let table_id = table_id(&fixture, "Residency").await; + fixture.data[0].0.drain().await.unwrap(); + remote + .provisioner + .admit_existing_partition("123456789012", &table_id, &[0; 16]) + .await + .unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let owner = authority + .load(fixture.data[0].0.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!( + owner.value().owner.as_ref().unwrap().session, + remote.session + ); + reads.reverse(); + expected.reverse(); + let result = sdk + .transact_get_items() + .set_transact_items(Some(reads)) + .send() + .await + .unwrap(); + assert_eq!(result.responses().len(), expected.len()); + for (response, item) in result.responses().iter().zip(&expected) { + assert_eq!(response.item(), Some(item)); + } + let after = authority + .load(fixture.data[0].0.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!( + after.value().root, + owner.value().root, + "large binary read must not publish participant mutations" + ); + // Rewrite through the returning PutItem path, whose declared input envelope + // can carry a legal item expanded by JSON escaping. + for item in &mut expected { + item.insert( + "payload".into(), + AwsAttributeValue::S("\0".repeat(380 * 1024)), + ); + sdk.put_item() + .table_name("Residency") + .set_item(Some(item.clone())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllOld) + .send() + .await + .unwrap(); + } + let before = authority + .load(fixture.data[0].0.cell_id()) + .await + .unwrap() + .unwrap(); + let reads = expected + .iter() + .map(|item| { + TransactGetItem::builder() + .get( + Get::builder() + .table_name("Residency") + .key("id", item["id"].clone()) + .build() + .unwrap(), + ) + .build() + }) + .collect::>(); + let read = sdk + .transact_get_items() + .set_transact_items(Some(reads)) + .send() + .await + .unwrap(); + for (response, item) in read.responses().iter().zip(&expected) { + assert_eq!(response.item(), Some(item)); + } + let after = authority + .load(fixture.data[0].0.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!( + after.value().root, + before.value().root, + "escaped string read must not publish participant mutations" + ); + remote.shutdown().await; + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_read_recovers_expired_directory_and_data_owners() { + for enabled in [false, true] { + read_recovers_expired_directory_and_data_owners(enabled).await; + } +} + +async fn read_recovers_expired_directory_and_data_owners(enabled: bool) { + println!("owner-expiry recovery: cache enabled={enabled}"); + let fixture = + Fixture::with_store_capacity_and_peer_cache(1, Arc::new(InMemory::new()), 8, enabled).await; let remote = Remote::new(&fixture).await; let sdk = sdk_without_retries(&fixture); let table_id = table_id(&fixture, "Residency").await; diff --git a/tests/peer_network/residency/recovery.rs b/tests/peer_network/residency/recovery.rs index e345b3c..7e9fc21 100644 --- a/tests/peer_network/residency/recovery.rs +++ b/tests/peer_network/residency/recovery.rs @@ -10,6 +10,163 @@ use cellule_runtime::{ const ACCOUNT: &str = "123456789012"; const ACCESS_KEY: &str = "AKIAIOSFODNN7EXAMPLE"; +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn recovered_index_can_release_residency_and_serve_later_sdk_requests() { + use futures_util::FutureExt; + let fixture = Fixture::with_capacity(1, 12).await; + CellAuthorizationStore::new(fixture.client.clone()) + .put_user_policy( + ACCOUNT, + "network-user", + "index-recovery", + &serde_json::json!({ + "Version": "2012-10-17", + "Statement": [{ + "Effect": "Allow", + "Action": "dynamodb:*", + "Resource": [ + "arn:aws:dynamodb:us-east-1:123456789012:table/ServingIndexFailover", + "arn:aws:dynamodb:us-east-1:123456789012:table/ServingIndexFailover/index/ByBucket" + ] + }] + }) + .to_string(), + ) + .await + .unwrap(); + let index = + crate::peer_network::global_indexes::IndexRecovery::create(&fixture.sdk, &fixture.client) + .await; + let storage = beyonddb::CellStorage::new(fixture.client.clone(), "us-east-1"); + storage + .project_index_changes(ACCOUNT, &index.targets[0], &index.table_id) + .await + .unwrap(); + index + .assert_settled(&fixture.sdk, &fixture.client, "before") + .await; + let authority = CellAuthority::new(fixture.layout.clone()); + let before = index.assert_owner(&authority, fixture.session).await; + // Release both original owners and restore their published roots. Retained + // incarnation plus an advanced epoch distinguishes recovery from bootstrap. + for target in &index.targets { + let proof = CellCatalog::new(fixture.layout.clone(), target.tenant()) + .lookup(target.cell_id()) + .await + .unwrap() + .unwrap(); + let control = authority.load(target.cell_id()).await.unwrap().unwrap(); + fixture + .node + .runtime() + .local_handle(proof, &control) + .await + .unwrap() + .unwrap() + .drain() + .await + .unwrap(); + } + fixture + .provisioner + .recover_registered_partitions(ACCOUNT, &fixture.client, &fixture.directory) + .await + .unwrap(); + index + .assert_settled(&fixture.sdk, &fixture.client, "before") + .await; + // Reclamation after the successful SDK query is legal and deterministic: + // force the index idle at the same post-query boundary as the CI failure. + let target = &index.targets[1]; + let proof = CellCatalog::new(fixture.layout.clone(), target.tenant()) + .lookup(target.cell_id()) + .await + .unwrap() + .unwrap(); + let control = authority.load(target.cell_id()).await.unwrap().unwrap(); + fixture + .node + .runtime() + .local_handle(proof, &control) + .await + .unwrap() + .unwrap() + .drain() + .await + .unwrap(); + let idle = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!(idle.value().state, ControlState::Idle); + assert!(idle.value().owner.is_none()); + assert!(idle.value().root.is_some()); + let checked = std::panic::AssertUnwindSafe(index.assert_recovered_authority( + &authority, + fixture.session, + &before, + )) + .catch_unwind() + .await; + // Reject a skipped takeover even when the existing image is readable. + let mut unchanged = Vec::new(); + for target in &index.targets { + unchanged.push( + authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .owner_fence(), + ); + } + let skipped = std::panic::AssertUnwindSafe(index.assert_recovered_authority( + &authority, + fixture.session, + &unchanged, + )) + .catch_unwind() + .await; + // A fresh bootstrap cannot stand in for restoring the original incarnation. + let mut bootstrapped = before.clone(); + bootstrapped[1].incarnation = + cellule_runtime::identity::IncarnationId::from_bytes(*uuid::Uuid::now_v7().as_bytes()); + let changed_incarnation = std::panic::AssertUnwindSafe(index.assert_recovered_authority( + &authority, + fixture.session, + &bootstrapped, + )) + .catch_unwind() + .await; + // A still-serving participant must belong to the expected replacement. + let wrong_owner = std::panic::AssertUnwindSafe(index.assert_recovered_authority( + &authority, + SessionId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + &before, + )) + .catch_unwind() + .await; + // The authority assertion must not require permanent residency. Prove the + // next signed write restores the same index and projects a newer image. + crate::peer_network::global_indexes::IndexRecovery::write(&fixture.sdk, "after").await; + storage + .project_index_changes(ACCOUNT, &index.targets[0], &index.table_id) + .await + .unwrap(); + index + .assert_settled(&fixture.sdk, &fixture.client, "after") + .await; + fixture.shutdown().await; + assert!(skipped.is_err(), "missed recovery must be rejected"); + assert!(changed_incarnation.is_err(), "bootstrap must be rejected"); + assert!( + wrong_owner.is_err(), + "unexpected serving owner must be rejected" + ); + assert!( + checked.is_ok(), + "post-recovery assertion rejected a published idle index" + ); +} + async fn interrupt_acquisition(fixture: &Fixture, handle: CellHandle) { let cell = handle.cell_id(); handle.drain().await.unwrap(); diff --git a/tests/peer_network/residency/saved_images.rs b/tests/peer_network/residency/saved_images.rs new file mode 100644 index 0000000..911d1c2 --- /dev/null +++ b/tests/peer_network/residency/saved_images.rs @@ -0,0 +1,540 @@ +//! Admission and immutable semantics of saved transaction images. + +use super::*; +use beyonddb::{ + GetItemInput, Json, PreparePartitionTransactionBounded, PreparePartitionTransactionInput, + PutItemInput, ReadTransactionInput, ResolvePartitionTransaction, ResolveTransactionInput, + TransactionOperation, +}; +use cellule_runtime::{MutationIdentity, identity::RequestId}; + +const ACCOUNT: &str = "123456789012"; + +fn identity() -> MutationIdentity { + let issued_at_ms = now_ms(); + MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + issued_at_ms, + expires_at_ms: issued_at_ms + 60_000, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn eight_saved_image_reads_fit_the_participant_mailbox() { + saved_image_mailbox_pressure(true).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn wide_saved_image_replies_reproduce_mailbox_pressure() { + saved_image_mailbox_pressure(false).await; +} + +async fn saved_image_mailbox_pressure(bounded: bool) { + use super::forward_cache::{CountedAuthority, CreationGate}; + use beyonddb::{ + BoundedTransactionReadResult, GetItemInput, ReadPartitionTransactionResult, + ReadPartitionTransactionResultBounded, ReadTransactionResultInput, TransactionReadResult, + }; + let store = Arc::new(CountedAuthority::default()); + let fixture = Fixture::with_store_capacity_and_peer_cache(1, store.clone(), 8, true).await; + let remote = super::provisioning::Remote::new(&fixture).await; + let client = remote.client_with_cache(&fixture, true); + let handle = &fixture.data[0].0; + let table_id = super::provisioning::table_id(&fixture, "Residency").await; + let range = super::cell_models::ranges(&fixture, "Residency") + .await + .remove(0); + let target = beyonddb::data_target(ACCOUNT, &table_id, &range.partition_id).unwrap(); + let coordinator_cell = *account_target(ACCOUNT).unwrap().cell_id().as_bytes(); + let AwsAttributeValue::S(key) = &fixture.data[0].1["id"] else { + panic!("string key"); + }; + let expected = Item::from([ + ("id".into(), AttributeValue::S(key.clone())), + ("value".into(), AttributeValue::S("committed".into())), + ]); + client + .command::( + &target, + identity(), + Json(PreparePartitionTransactionInput { + table_id: table_id.clone(), + epoch: range.epoch, + transaction_id: [220; 16], + coordinator_cell, + coordinator_key: vec![220; 16], + operations: vec![TransactionOperation::Read(GetItemInput { + table_name: "Residency".into(), + table_id: table_id.clone(), + key: Item::from([("id".into(), AttributeValue::S(key.clone()))]), + })], + }), + ) + .await + .unwrap(); + client + .command::( + &target, + identity(), + Json(ResolveTransactionInput { + transaction_id: [220; 16], + coordinator_cell, + commit: true, + }), + ) + .await + .unwrap(); + let input = Json(ReadTransactionResultInput { + transaction: ReadTransactionInput { + transaction_id: [220; 16], + coordinator_cell, + }, + position: 0, + }); + // Hold a preceding real publication: FIFO snapshot queries must retain + // their reply reservations until it completes, just as under slow storage. + let gate = CreationGate { + paths: std::collections::HashSet::from([fixture + .layout + .control_path(handle.cell_id().as_bytes())]), + entered: Arc::new(tokio::sync::Semaphore::new(0)), + release: Arc::new(tokio::sync::Semaphore::new(0)), + }; + *store.publication_gate.lock().unwrap() = Some(gate.clone()); + let blocking_client = client.clone(); + let blocking_target = target.clone(); + let blocked = tokio::spawn(async move { + blocking_client + .command::( + &blocking_target, + identity(), + Json(PreparePartitionTransactionInput { + table_id: table_id.clone(), + epoch: range.epoch, + transaction_id: [221; 16], + coordinator_cell, + coordinator_key: vec![221; 16], + operations: vec![TransactionOperation::Put(PutItemInput { + table_name: "Residency".into(), + table_id, + item: Item::from([( + "id".into(), + AttributeValue::S("held-image-publication".into()), + )]), + condition: None, + })], + }), + ) + .await + }); + tokio::time::timeout(std::time::Duration::from_secs(5), gate.entered.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let before = fixture.node.runtime().stats().retained_bytes(); + let mut tasks = Vec::new(); + for _ in 0..8 { + let client = client.clone(); + let target = target.clone(); + let input = input.clone(); + tasks.push(tokio::spawn(async move { + if bounded { + client + .query::(&target, None, input) + .await + .map(|result| result.output) + .map_err(|error| format!("{error:?}")) + } else { + client + .query::(&target, None, input) + .await + .map(|result| match result.output.0 { + TransactionReadResult::Unavailable => { + BoundedTransactionReadResult::Unavailable + } + TransactionReadResult::Item(item) => { + BoundedTransactionReadResult::Item(item) + } + }) + .map_err(|error| format!("{error:?}")) + } + })); + } + let observed = tokio::time::timeout(std::time::Duration::from_secs(10), async { + loop { + let admitted = if bounded { + // Each held request reserves a 64 KiB reply and a 64 KiB + // peer buffer. With these tiny inputs, eight peer buffers and + // only seven reply reservations stay below this threshold. + fixture.node.runtime().stats().retained_bytes() >= before + 8 * 128 * 1024 + && tasks.iter().all(|task| !task.is_finished()) + } else { + tasks.iter().filter(|task| task.is_finished()).count() >= 5 + }; + if admitted { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .is_ok(); + // Wait for refusals to become visible after the first wide reservation. + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + let retained = fixture.node.runtime().stats().retained_bytes(); + *store.publication_gate.lock().unwrap() = None; + gate.release.add_permits(64); + blocked.await.unwrap().unwrap(); + let mut accepted = 0; + let mut refused = Vec::new(); + for task in tasks { + match tokio::time::timeout(std::time::Duration::from_secs(30), task) + .await + .unwrap() + .unwrap() + { + Ok(result) => { + assert_eq!( + result, + BoundedTransactionReadResult::Item(Some(expected.clone())) + ); + accepted += 1; + } + Err(error) => refused.push(error), + } + } + client + .command::( + &target, + identity(), + Json(ResolveTransactionInput { + transaction_id: [221; 16], + coordinator_cell, + commit: false, + }), + ) + .await + .unwrap(); + remote.shutdown().await; + fixture.shutdown().await; + println!( + "held saved-image queries: bounded={bounded}, admitted={accepted}, observed={observed}, retained={before}->{retained}, refused={refused:?}" + ); + assert!(observed); + assert_eq!( + accepted, + if bounded { 8 } else { 3 }, + "snapshot reply reservations refused: {refused:?}" + ); + if bounded { + assert!(retained - before < 8 * 140 * 1024); + assert!(refused.is_empty()); + } else { + assert!(retained - before >= 12 * 1024 * 1024); + assert_eq!(refused.len(), 5); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn saved_images_preserve_identity_wide_items_and_owner_restoration() { + use beyonddb::{ + BoundedTransactionReadResult, ParticipantTransactionState, ReadPartitionTransaction, + ReadPartitionTransactionResult, ReadPartitionTransactionResultBounded, + ReadTransactionResultInput, ReleasePartitionTransactionReads, TransactionReadResult, + }; + let fixture = Fixture::with_capacity(1, 16).await; + let sdk = super::provisioning::sdk_without_retries(&fixture); + let small = fixture.data[0].1.clone(); + let mut expected = vec![Some(small.clone())]; + for (id, payload) in [ + ( + "saved-binary", + AwsAttributeValue::B(vec![0xa5; 380 * 1024].into()), + ), + ( + "saved-escaped", + AwsAttributeValue::S("\0".repeat(380 * 1024)), + ), + ] { + let item = SdkItem::from([ + ("id".into(), AwsAttributeValue::S(id.into())), + ("payload".into(), payload), + ]); + sdk.put_item() + .table_name("Residency") + .set_item(Some(item.clone())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllOld) + .send() + .await + .unwrap(); + expected.push(Some(item)); + } + expected.push(None); + let table_id = super::provisioning::table_id(&fixture, "Residency").await; + let range = super::cell_models::ranges(&fixture, "Residency") + .await + .remove(0); + let target = beyonddb::data_target(ACCOUNT, &table_id, &range.partition_id).unwrap(); + let coordinator_cell = *account_target(ACCOUNT).unwrap().cell_id().as_bytes(); + let transaction = ReadTransactionInput { + transaction_id: [224; 16], + coordinator_cell, + }; + let operations = expected + .iter() + .map(|item| { + let id = item + .as_ref() + .map(|item| item["id"].clone()) + .unwrap_or_else(|| AwsAttributeValue::S("saved-absent".into())); + let AwsAttributeValue::S(id) = id else { + panic!("string key"); + }; + TransactionOperation::Read(GetItemInput { + table_name: "Residency".into(), + table_id: table_id.clone(), + key: Item::from([("id".into(), AttributeValue::S(id))]), + }) + }) + .collect(); + fixture + .client + .command::( + &target, + identity(), + Json(PreparePartitionTransactionInput { + table_id: table_id.clone(), + epoch: range.epoch, + transaction_id: transaction.transaction_id, + coordinator_cell, + coordinator_key: transaction.transaction_id.to_vec(), + operations, + }), + ) + .await + .unwrap(); + let input = |position| { + Json(ReadTransactionResultInput { + transaction: transaction.clone(), + position, + }) + }; + assert_eq!( + fixture + .client + .query::(&target, None, input(0)) + .await + .unwrap() + .output, + BoundedTransactionReadResult::Unavailable + ); + fixture + .client + .command::( + &target, + identity(), + Json(ResolveTransactionInput { + transaction_id: transaction.transaction_id, + coordinator_cell, + commit: true, + }), + ) + .await + .unwrap(); + // A saved image is independent of later writes to the live item. + let mut changed = small; + changed.insert("value".into(), AwsAttributeValue::S("newer".into())); + sdk.put_item() + .table_name("Residency") + .set_item(Some(changed)) + .send() + .await + .unwrap(); + let remote = super::provisioning::Remote::new(&fixture).await; + for restored in [false, true] { + if restored { + fixture.data[0].0.drain().await.unwrap(); + remote + .provisioner + .admit_existing_partition(ACCOUNT, &table_id, &range.partition_id) + .await + .unwrap(); + } + let client = remote.client_with_cache(&fixture, true); + for (position, item) in expected.iter().enumerate() { + let request = input(u8::try_from(position).unwrap()); + let bounded = client + .query::(&target, None, request.clone()) + .await + .unwrap() + .output; + let wide = client + .query::(&target, None, request) + .await + .unwrap() + .output + .0; + let core_item = item.as_ref().map(|item| { + item.iter() + .map(|(name, value)| { + let value = match value { + AwsAttributeValue::S(value) => AttributeValue::S(value.clone()), + AwsAttributeValue::B(value) => { + AttributeValue::B(value.as_ref().to_vec()) + } + value => panic!("unexpected fixture attribute: {value:?}"), + }; + (name.clone(), value) + }) + .collect::() + }); + assert_eq!(wide, TransactionReadResult::Item(core_item.clone())); + if matches!(position, 1 | 2) { + assert_eq!(bounded, BoundedTransactionReadResult::WideRequired); + } else { + assert_eq!(bounded, BoundedTransactionReadResult::Item(core_item)); + } + } + for request in [ + input(4), + Json(ReadTransactionResultInput { + transaction: ReadTransactionInput { + coordinator_cell: [225; 32], + ..transaction.clone() + }, + position: 0, + }), + ] { + assert_eq!( + client + .query::(&target, None, request) + .await + .unwrap() + .output, + BoundedTransactionReadResult::Unavailable + ); + } + } + let client = remote.client_with_cache(&fixture, true); + client + .command::(&target, identity(), Json(transaction.clone())) + .await + .unwrap(); + assert_eq!( + client + .query::(&target, None, input(0)) + .await + .unwrap() + .output, + BoundedTransactionReadResult::Unavailable + ); + assert_eq!( + client + .query::(&target, None, Json(transaction)) + .await + .unwrap() + .output + .0, + ParticipantTransactionState::Committed + ); + remote.shutdown().await; + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn signed_cross_cell_saved_images_and_wide_fallback_survive_owner_change() { + use aws_sdk_dynamodb::types::{Get, TransactGetItem}; + use cellule_runtime::control::authority::CellAuthority; + let fixture = Fixture::with_capacity(2, 16).await; + let sdk = super::provisioning::sdk_without_retries(&fixture); + let mut expected = fixture + .data + .iter() + .map(|(_, item)| Some(item.clone())) + .collect::>(); + let mut keys = expected + .iter() + .map(|item| item.as_ref().unwrap()["id"].clone()) + .collect::>(); + keys.push(AwsAttributeValue::S("saved-sdk-absent".into())); + expected.push(None); + let reads = keys + .into_iter() + .map(|id| { + TransactGetItem::builder() + .get( + Get::builder() + .table_name("Residency") + .key("id", id) + .build() + .unwrap(), + ) + .build() + }) + .rev() + .collect::>(); + for wide in [false, true] { + if wide { + for (position, item) in expected.iter_mut().flatten().enumerate() { + item.insert( + "payload".into(), + if position == 0 { + AwsAttributeValue::B(vec![0xa5; 380 * 1024].into()) + } else { + AwsAttributeValue::S("\0".repeat(380 * 1024)) + }, + ); + sdk.put_item() + .table_name("Residency") + .set_item(Some(item.clone())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllOld) + .send() + .await + .unwrap(); + } + } + let read = sdk + .transact_get_items() + .set_transact_items(Some(reads.clone())) + .send() + .await + .unwrap(); + assert_eq!(read.responses().len(), expected.len()); + for (result, item) in read.responses().iter().zip(expected.iter().rev()) { + assert_eq!(result.item(), item.as_ref()); + } + } + let remote = super::provisioning::Remote::new(&fixture).await; + let table_id = super::provisioning::table_id(&fixture, "Residency").await; + let ranges = super::cell_models::ranges(&fixture, "Residency").await; + let authority = CellAuthority::new(fixture.layout.clone()); + for ((handle, _), range) in fixture.data.iter().zip(&ranges) { + let before = authority.load(handle.cell_id()).await.unwrap().unwrap(); + handle.drain().await.unwrap(); + remote + .provisioner + .admit_existing_partition(ACCOUNT, &table_id, &range.partition_id) + .await + .unwrap(); + let restored = authority.load(handle.cell_id()).await.unwrap().unwrap(); + assert_eq!(restored.value().incarnation, before.value().incarnation); + assert!(restored.value().epoch > before.value().epoch); + assert_eq!( + restored.value().owner.as_ref().unwrap().session, + remote.session + ); + } + let restored = sdk + .transact_get_items() + .set_transact_items(Some(reads)) + .send() + .await + .unwrap(); + assert_eq!(restored.responses().len(), expected.len()); + for (result, item) in restored.responses().iter().zip(expected.iter().rev()) { + assert_eq!(result.item(), item.as_ref()); + } + remote.shutdown().await; + fixture.shutdown().await; +} diff --git a/tests/peer_network/residency/update_batch.rs b/tests/peer_network/residency/update_batch.rs new file mode 100644 index 0000000..65155cd --- /dev/null +++ b/tests/peer_network/residency/update_batch.rs @@ -0,0 +1,323 @@ +use super::provisioning::{create, sdk_without_retries, table_id}; +use super::*; +use beyonddb::{ + BatchedPartitionUpdate, PartitionUpdateBatch, PartitionUpdateBatchOutcome, + PartitionUpdateInput, PartitionUpdateOutcome, +}; +use cellule_runtime::client::InvocationError; +use cellule_runtime::identity::RequestId; +use extenddb_core::expression::{CompareOp, Expr, ExpressionMaps, PathElement, UpdateAction}; +use extenddb_storage::StreamEngine; + +const ACCOUNT: &str = "123456789012"; + +fn identity() -> cellule_runtime::MutationIdentity { + let now = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_millis() as i64; + cellule_runtime::MutationIdentity { + request_id: RequestId::from_bytes(*uuid::Uuid::now_v7().as_bytes()), + issued_at_ms: now, + expires_at_ms: now + 60_000, + } +} + +fn add(table: &str, epoch: u64, key: &str, condition: bool) -> BatchedPartitionUpdate { + let path = vec![PathElement::Attribute("value".into())]; + let actions = [UpdateAction::Add { + path: path.clone(), + value: Expr::Placeholder("one".into()), + }]; + let condition = condition.then(|| Expr::Compare { + left: Box::new(Expr::Path(path)), + op: CompareOp::Eq, + right: Box::new(Expr::Placeholder("one".into())), + }); + let maps = ExpressionMaps::new( + HashMap::new(), + HashMap::from([("one".into(), AttributeValue::N("1".into()))]), + ); + BatchedPartitionUpdate { + input: PartitionUpdateInput::from_expression( + table.into(), + epoch, + Item::from([("id".into(), AttributeValue::S(key.into()))]), + &actions, + condition.as_ref(), + &maps, + ), + return_old: true, + return_new: true, + } +} + +async fn records(fixture: &Fixture, arn: &str) -> Vec { + let streams = beyonddb::CellStorage::new(fixture.client.clone(), "us-east-1"); + let description = streams + .describe_stream( + ACCOUNT, + &extenddb_core::types::DescribeStreamInput { + stream_arn: arn.into(), + limit: Some(100), + exclusive_start_shard_id: None, + }, + ) + .await + .unwrap(); + let shard = &description.shards[0].shard_id; + streams.validate_shard(ACCOUNT, arn, shard).await.unwrap(); + streams + .get_stream_records(ACCOUNT, shard, None, 100) + .await + .unwrap() + .0 +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn returned_update_batch_isolates_conditions_and_keeps_stream_ordinals() { + let fixture = Fixture::with_capacity(1, 16).await; + let sdk = sdk_without_retries(&fixture); + let table = "ResidencyReturnedStream"; + let arn = create(&sdk, table, false) + .stream_specification( + aws_sdk_dynamodb::types::StreamSpecification::builder() + .stream_enabled(true) + .stream_view_type(aws_sdk_dynamodb::types::StreamViewType::NewAndOldImages) + .build() + .unwrap(), + ) + .send() + .await + .unwrap() + .table_description + .unwrap() + .latest_stream_arn + .unwrap(); + let id = table_id(&fixture, table).await; + let range = super::cell_models::ranges(&fixture, table).await.remove(0); + let target = beyonddb::data_target(ACCOUNT, &id, &range.partition_id).unwrap(); + sdk.put_item() + .table_name(table) + .item("id", AwsAttributeValue::S("condition".into())) + .item("value", AwsAttributeValue::N("0".into())) + .send() + .await + .unwrap(); + let error = sdk + .update_item() + .table_name(table) + .key("id", AwsAttributeValue::S("condition".into())) + .update_expression("ADD #value :one") + .condition_expression("#value = :one") + .expression_attribute_names("#value", "value") + .expression_attribute_values(":one", AwsAttributeValue::N("1".into())) + .return_values_on_condition_check_failure( + aws_sdk_dynamodb::types::ReturnValuesOnConditionCheckFailure::AllOld, + ) + .send() + .await + .unwrap_err(); + let aws_sdk_dynamodb::operation::update_item::UpdateItemError::ConditionalCheckFailedException( + error, + ) = error.as_service_error().unwrap() + else { + panic!("incorrect SDK condition error"); + }; + assert_eq!( + error.item().unwrap()["value"], + AwsAttributeValue::N("0".into()) + ); + let updates = vec![ + add(&id, range.epoch, "first", false), + add(&id, range.epoch, "condition", true), + add(&id, range.epoch, "second", false), + ]; + let request = identity(); + let result = fixture + .client + .command::(&target, request, Json(updates.clone())) + .await + .unwrap(); + let PartitionUpdateBatchOutcome::Results(results) = result.output else { + panic!("unexpected batch fallback"); + }; + assert!(matches!( + results[0], + PartitionUpdateOutcome::Applied { old: None, .. } + )); + assert!(matches!( + results[2], + PartitionUpdateOutcome::Applied { old: None, .. } + )); + let PartitionUpdateOutcome::ConditionFailed(Some(old)) = &results[1] else { + panic!("condition result lost its old image"); + }; + assert_eq!(old["value"], AttributeValue::N("0".into())); + let before = records(&fixture, &arn).await; + assert_eq!( + before.len(), + 3, + "each successful mutation needs its own stream record" + ); + assert_ne!(before[1].event_id, before[2].event_id); + fixture + .client + .command::(&target, request, Json(updates)) + .await + .unwrap(); + assert_eq!( + records(&fixture, &arn).await.len(), + 3, + "receipt replay must not repeat stream intent" + ); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + for (key, value) in [("first", "1"), ("condition", "0"), ("second", "1")] { + let item = sdk + .get_item() + .table_name(table) + .key("id", AwsAttributeValue::S(key.into())) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::N(value.into())); + } + assert_eq!(records(&fixture, &arn).await.len(), 3); + // A large stored image makes even a tiny ADD exceed the compact batch + // result bound. Its rejected receipt must roll back both item and stream. + sdk.put_item() + .table_name(table) + .item("id", AwsAttributeValue::S("large".into())) + .item("value", AwsAttributeValue::N("0".into())) + .item("padding", AwsAttributeValue::S("x".repeat(140_000))) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllOld) + .send() + .await + .unwrap(); + let rejected = fixture + .client + .command::( + &target, + identity(), + Json(vec![add(&id, range.epoch, "large", false)]), + ) + .await + .unwrap_err(); + assert!( + matches!(rejected, InvocationError::Rejected(ref committed) if committed.output == PartitionUpdateBatchOutcome::IndividualRequired) + ); + let item = sdk + .get_item() + .table_name(table) + .key("id", AwsAttributeValue::S("large".into())) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(item["value"], AwsAttributeValue::N("0".into())); + assert_eq!( + records(&fixture, &arn).await.len(), + 4, + "rolled-back ADD must emit no stream record" + ); + sdk.update_item() + .table_name(table) + .key("id", AwsAttributeValue::S("large".into())) + .update_expression("ADD #value :one") + .expression_attribute_names("#value", "value") + .expression_attribute_values(":one", AwsAttributeValue::N("1".into())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllNew) + .send() + .await + .unwrap(); + let item = sdk + .get_item() + .table_name(table) + .key("id", AwsAttributeValue::S("large".into())) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!( + item["value"], + AwsAttributeValue::N("1".into()), + "fallback must apply ADD once" + ); + assert_eq!(records(&fixture, &arn).await.len(), 5); + fixture.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sdk_returned_update_compacts_large_escaped_old_and_new_images() { + let fixture = Fixture::with_capacity(1, 16).await; + let sdk = sdk_without_retries(&fixture); + let padding = "\0".repeat(390_000); + sdk.put_item() + .table_name("Residency") + .item("id", AwsAttributeValue::S("escaped".into())) + .item("value", AwsAttributeValue::N("0".into())) + .item("padding", AwsAttributeValue::S(padding.clone())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllOld) + .send() + .await + .unwrap(); + let result = sdk + .update_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("escaped".into())) + .update_expression("ADD #value :one") + .expression_attribute_names("#value", "value") + .expression_attribute_values(":one", AwsAttributeValue::N("1".into())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllOld) + .send() + .await + .unwrap(); + let old = result.attributes.unwrap(); + assert_eq!(old["value"], AwsAttributeValue::N("0".into())); + assert_eq!(old["padding"], AwsAttributeValue::S(padding.clone())); + let id = table_id(&fixture, "Residency").await; + let range = super::cell_models::ranges(&fixture, "Residency") + .await + .remove(0); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap() + .drain() + .await + .unwrap(); + fixture + .provisioner + .admit_existing_partition(ACCOUNT, &id, &range.partition_id) + .await + .unwrap(); + let new = sdk + .get_item() + .table_name("Residency") + .key("id", AwsAttributeValue::S("escaped".into())) + .send() + .await + .unwrap() + .item + .unwrap(); + assert_eq!(new["value"], AwsAttributeValue::N("1".into())); + assert_eq!(new["padding"], AwsAttributeValue::S(padding)); + fixture.shutdown().await; +} diff --git a/tests/server_binary.rs b/tests/server_binary.rs index e9a6e8d..4bd5abd 100644 --- a/tests/server_binary.rs +++ b/tests/server_binary.rs @@ -5,6 +5,7 @@ mod support; mod server_binary { mod capacity; mod coordinator_recovery; + mod follower_durability; pub(super) mod global_indexes; mod large_reads; pub(super) mod local_indexes; @@ -210,13 +211,32 @@ fn stop(child: &mut Child, log: &Path) { struct RustfsContainer { name: String, log: PathBuf, + data_dir: Option, } impl RustfsContainer { fn start(address: SocketAddr, log: PathBuf) -> Self { + let name = format!("beyonddb-test-{}", uuid::Uuid::now_v7()); + // Colima can share the host home while its own volume disk is full. + // Each test owns a fresh child of the explicitly selected bind root. + let data_dir = std::env::var_os("BEYONDDB_TEST_RUSTFS_BIND_ROOT").map(|root| { + let path = PathBuf::from(root).join(&name); + fs::create_dir_all(&path).unwrap(); + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + fs::set_permissions(&path, fs::Permissions::from_mode(0o777)).unwrap(); + } + path + }); + let volume = data_dir.as_ref().map_or_else( + || "/data".to_owned(), + |path| format!("{}:/data", path.display()), + ); let container = Self { - name: format!("beyonddb-test-{}", uuid::Uuid::now_v7()), + name, log, + data_dir, }; run(Command::new("docker").args([ "run", @@ -234,7 +254,7 @@ impl RustfsContainer { "--env", "RUSTFS_OBS_LOG_DIRECTORY=/data/logs", "--volume", - "/data", + &volume, "ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858", ])); container @@ -261,11 +281,20 @@ impl Drop for RustfsContainer { // Failed recovery fixtures retain their object data for startup // replay. Stop serving, but preserve the named container and volume. eprintln!("retained RustFS container for replay: {}", self.name); + if let Some(path) = &self.data_dir { + eprintln!("retained RustFS bind data: {}", path.display()); + } command.args(["stop", &self.name]); } else { command.args(["rm", "--force", "--volumes", &self.name]); } - let _ = command.output(); + let removed = command.output().is_ok_and(|output| output.status.success()); + if removed + && !std::thread::panicking() + && let Some(path) = &self.data_dir + { + let _ = fs::remove_dir_all(path); + } } } diff --git a/tests/server_binary/follower_durability.rs b/tests/server_binary/follower_durability.rs new file mode 100644 index 0000000..3383d22 --- /dev/null +++ b/tests/server_binary/follower_durability.rs @@ -0,0 +1,619 @@ +use crate::*; + +use std::sync::{Arc, Mutex}; + +use axum::{body::Body, extract::State, http::Request, response::Response}; +use beyonddb::{APPLICATION_ID, Beyonddb}; +use cellule_app::CellApplication; +use cellule_peer_http::{LoadedPeerTls, PeerHttpRoundTrip}; +use cellule_runtime::{ + Digest, + client::CellClient, + control::authority::CellAuthority, + ltx::CellStorageLayout, + node::NodeDirectory, + peer::{PeerPrincipal, PeerSigner}, + registry::BuildDescriptor, +}; +use cellule_store::Store; +use tokio_util::sync::CancellationToken; + +#[derive(Clone)] +struct PublicationBarrier { + upstream: SocketAddr, + client: reqwest::Client, + cell: Arc>>, + blocked: CancellationToken, + release: CancellationToken, +} + +async fn proxy(State(barrier): State, request: Request) -> Response { + let (parts, body) = request.into_parts(); + let path = parts.uri.path_and_query().unwrap().as_str(); + let blocked = matches!(parts.method.as_str(), "PUT" | "POST") + && barrier + .cell + .lock() + .unwrap() + .as_ref() + .is_some_and(|cell| path.contains(cell) && path.contains("/objects/")); + let body = axum::body::to_bytes(body, 64 << 20).await.unwrap(); + if blocked { + barrier.blocked.cancel(); + barrier.release.cancelled().await; + } + // Preserve the signed Host, URI, headers, and body while changing only the + // connection destination. RustFS still validates the original SigV4 request. + let response = barrier + .client + .request(parts.method, format!("http://{}{path}", barrier.upstream)) + .headers(parts.headers) + .body(body) + .send() + .await + .unwrap(); + let status = response.status(); + let headers = response.headers().clone(); + let bytes = response.bytes().await.unwrap(); + let mut result = Response::new(Body::from(bytes)); + *result.status_mut() = status; + *result.headers_mut() = headers; + result +} + +fn issue_certificate(root: &Path, name: &str) { + run(Command::new("openssl") + .args(["genpkey", "-algorithm", "ED25519", "-out"]) + .arg(root.join(format!("{name}.key")))); + run(Command::new("openssl") + .args(["req", "-new", "-subj", "/CN=localhost", "-key"]) + .arg(root.join(format!("{name}.key"))) + .arg("-out") + .arg(root.join(format!("{name}.csr")))); + run(Command::new("openssl") + .args(["x509", "-req", "-days", "1", "-CAcreateserial", "-in"]) + .arg(root.join(format!("{name}.csr"))) + .arg("-CA") + .arg(root.join("ca.crt")) + .arg("-CAkey") + .arg(root.join("ca.key")) + .arg("-extfile") + .arg(root.join("peer.ext")) + .arg("-out") + .arg(root.join(format!("{name}.crt")))); +} + +fn now_ms() -> i64 { + i64::try_from( + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "requires Docker, aws CLI, and the pinned RustFS GA image"] +async fn acknowledged_untiered_item_survives_owner_process_kill() { + let mut fixture = process_fixture(1).await; + stop(&mut fixture.child, &fixture.log); + let root = fixture.root.path(); + let barrier = PublicationBarrier { + upstream: fixture.s3, + client: reqwest::Client::new(), + cell: Arc::new(Mutex::new(None)), + blocked: CancellationToken::new(), + release: CancellationToken::new(), + }; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let proxy_address = listener.local_addr().unwrap(); + let router = axum::Router::new() + .fallback(proxy) + .with_state(barrier.clone()); + let proxy_server = tokio::spawn(async move { + axum::serve(listener, router).await.unwrap(); + }); + let mut config: serde_json::Value = + serde_json::from_slice(&fs::read(&fixture.config).unwrap()).unwrap(); + config["follower_store_bytes"] = json!(1_u64 << 30); + config["follower_durability_enabled"] = json!(true); + config["auth_cache_enabled"] = json!(true); + config["runtime_metrics_file"] = json!(root.join("owner-metrics.json")); + fs::write(&fixture.config, config.to_string()).unwrap(); + let mut followers = Vec::new(); + for name in ["follower-a", "follower-b"] { + issue_certificate(root, name); + let peer = free_addr(); + let public = free_addr(); + let mut next = config.clone(); + next["node_id"] = json!(uuid::Uuid::now_v7()); + next["data_dir"] = json!(root.join(name)); + next["peer_bind"] = json!(peer); + next["peer_endpoint"] = json!(format!("https://{peer}")); + next["public_bind"] = json!(public); + next["public_endpoint"] = json!(format!("http://{public}")); + next["peer_certificate"] = json!(root.join(format!("{name}.crt"))); + next["peer_private_key"] = json!(root.join(format!("{name}.key"))); + next["owned_accounts"] = json!([]); + next["owned_access_keys"] = json!([]); + next["bootstrap"] = serde_json::Value::Null; + next["runtime_metrics_file"] = json!(root.join(format!("{name}-metrics.json"))); + let path = root.join(format!("{name}.json")); + let log = root.join(format!("{name}.log")); + fs::write(&path, next.to_string()).unwrap(); + let child = start(&path, &log, false, proxy_address); + followers.push((child, path, log, peer, public)); + } + fixture.child = start(&fixture.config, &fixture.log, false, proxy_address); + wait_healthy(&mut fixture.child, fixture.public, &fixture.log); + for (child, _, log, _, public) in &mut followers { + wait_healthy(child, *public, log); + } + let tls = LoadedPeerTls::load( + &root.join("peer.crt"), + &root.join("peer.key"), + &root.join("ca.crt"), + "localhost", + ) + .unwrap(); + let application = Beyonddb::compile(BuildDescriptor { + source_revision: option_env!("BEYONDDB_SOURCE_REVISION") + .map(str::to_owned) + .unwrap_or_else(|| { + blake3::hash(include_bytes!("../../src/bin/beyonddb.rs")) + .to_hex() + .to_string() + }), + cargo_lock_digest: Digest::from_bytes( + *blake3::hash(include_bytes!("../../Cargo.lock")).as_bytes(), + ), + }) + .unwrap(); + let store = object_store::aws::AmazonS3Builder::new() + .with_bucket_name("beyonddb-test") + .with_region("us-east-1") + .with_endpoint(format!("http://{}", fixture.s3)) + .with_allow_http(true) + .with_access_key_id("crab") + .with_secret_access_key("crab") + .build() + .unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(store)), + object_store::path::Path::from("beyonddb"), + *APPLICATION_ID.as_bytes(), + ); + let authority = CellAuthority::new(layout.clone()); + let directory = NodeDirectory::new( + layout.clone(), + tls.fleet(), + application.descriptor_digest(), + application.registry().release_digest(), + ); + let created = fixture + .sdk + .create_table() + .table_name("FollowerRecovery") + .key_schema( + aws_sdk_dynamodb::types::KeySchemaElement::builder() + .attribute_name("id") + .key_type(aws_sdk_dynamodb::types::KeyType::Hash) + .build() + .unwrap(), + ) + .attribute_definitions( + aws_sdk_dynamodb::types::AttributeDefinition::builder() + .attribute_name("id") + .attribute_type(aws_sdk_dynamodb::types::ScalarAttributeType::S) + .build() + .unwrap(), + ) + .billing_mode(aws_sdk_dynamodb::types::BillingMode::PayPerRequest) + .stream_specification( + aws_sdk_dynamodb::types::StreamSpecification::builder() + .stream_enabled(true) + .stream_view_type(aws_sdk_dynamodb::types::StreamViewType::KeysOnly) + .build() + .unwrap(), + ) + .send() + .await + .unwrap(); + let table = created.table_description().unwrap().table_id().unwrap(); + let arn = created + .table_description() + .unwrap() + .latest_stream_arn() + .unwrap(); + let target = beyonddb::data_target("123456789012", table, &[0; 16]).unwrap(); + let deadline = Instant::now() + Duration::from_secs(30); + let owner = loop { + fixture + .sdk + .put_item() + .table_name("FollowerRecovery") + .item("id", AttributeValue::S("warmup".into())) + .send() + .await + .unwrap(); + let current = authority.load(target.cell_id()).await.unwrap().unwrap(); + let owner = current.value().owner.as_ref().unwrap().clone(); + let node = directory + .load(owner.session, now_ms()) + .await + .unwrap() + .unwrap(); + // Enrollment is sufficient to submit the first cut. Normal warmup + // writes can all win object publication and leave the log inactive. + // The withheld write below must force fsync and authoritative activation. + if node.advertisement().log().is_some() { + break owner; + } + if Instant::now() >= deadline { + match directory.live(now_ms(), 16).await { + Ok(live) => { + let capacity = live + .iter() + .map(|node| (node.node(), node.session(), node.capacity(), node.log())) + .collect::>(); + panic!("owner never enrolled a follower log; live node capacity: {capacity:?}"); + } + Err(error) => { + panic!("owner never enrolled a follower log; directory read failed: {error}"); + } + } + } + tokio::time::sleep(Duration::from_millis(100)).await; + }; + // Obtain the owner's FIFO observation after the last warmup command, then + // wait for that exact committed sequence to reach object storage. An SDK + // acknowledgement may use follower durability while publication is pending. + let local = directory + .live(now_ms(), 16) + .await + .unwrap() + .into_iter() + .find(|node| node.endpoint() == format!("https://{}", fixture.peer)) + .unwrap(); + let session = local.session(); + let hex = |bytes: &[u8]| { + bytes + .iter() + .map(|byte| format!("{byte:02x}")) + .collect::() + }; + let client = CellClient::peer( + application.registry(), + Arc::new(PeerSigner::new( + session, + application.registry().release_digest(), + tls.signing_key().clone(), + )), + PeerPrincipal { + issuer: format!("beyonddb-peer:{}", hex(directory.fleet().as_bytes())), + subject: hex(session.as_bytes()), + actions: vec!["beyonddb.cell.invoke".into()], + }, + Arc::new(PeerHttpRoundTrip::new( + Arc::new(beyonddb::BeyonddbPeerScope), + authority.clone(), + directory.clone(), + Arc::new(tls.client_identity()), + session, + )), + ); + let warmup = client + .query::(&target, None, beyonddb::Json(())) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(30), async { + loop { + let published = authority.load(target.cell_id()).await.unwrap().unwrap(); + if published + .value() + .root + .as_ref() + .is_some_and(|root| root.commit_sequence >= warmup.receipt.commit_sequence) + { + break; + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + }) + .await + .expect("all warmup commits must publish before the object barrier is installed"); + let before = authority.load(target.cell_id()).await.unwrap().unwrap(); + let predecessor = before.value().root.as_ref().unwrap().commit_sequence; + let cell = target + .cell_id() + .as_bytes() + .iter() + .map(|byte| format!("{byte:02x}")) + .collect::(); + *barrier.cell.lock().unwrap() = Some(format!("/cells/{cell}/inc/")); + let mut item = HashMap::from([ + ("id".into(), AttributeValue::S("acknowledged".into())), + ( + "value".into(), + AttributeValue::S("survived-process-kill".into()), + ), + ]); + let sdk = aws_sdk_dynamodb::Client::from_conf( + fixture + .sdk + .config() + .to_builder() + .retry_config(aws_sdk_dynamodb::config::retry::RetryConfig::disabled()) + .build(), + ); + tokio::time::timeout( + Duration::from_secs(5), + sdk.put_item() + .table_name("FollowerRecovery") + .set_item(Some(item.clone())) + .send(), + ) + .await + .expect("follower proof must acknowledge while object publication is withheld") + .unwrap(); + let active = directory + .load(owner.session, now_ms()) + .await + .unwrap() + .unwrap(); + assert!( + active.advertisement().log().is_some_and(|log| log.active()), + "the first withheld write must activate the enrolled follower log before acknowledgement" + ); + tokio::time::timeout(Duration::from_secs(5), async { + let metrics = match followers + .iter() + .position(|(_, _, _, peer, _)| owner.endpoint == format!("https://{peer}")) + { + Some(index) => root.join(format!( + "follower-{}-metrics.json", + if index == 0 { "a" } else { "b" } + )), + None => root.join("owner-metrics.json"), + }; + loop { + if let Ok(bytes) = fs::read(&metrics) + && let Ok(snapshot) = serde_json::from_slice::(&bytes) + && snapshot["command_responses"]["fleet"]["count"] + .as_u64() + .is_some_and(|count| count > 0) + { + assert_eq!(snapshot["version"], 1); + break; + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + }) + .await + .expect("serving binary must write follower response metrics"); + let updated = sdk + .update_item() + .table_name("FollowerRecovery") + .key("id", item["id"].clone()) + .update_expression("SET #version = :version") + .expression_attribute_names("#version", "version") + .expression_attribute_values(":version", AttributeValue::N("1".into())) + .return_values(aws_sdk_dynamodb::types::ReturnValue::AllNew) + .send() + .await + .unwrap(); + item.insert("version".into(), AttributeValue::N("1".into())); + assert_eq!(updated.attributes(), Some(&item)); + sdk.delete_item() + .table_name("FollowerRecovery") + .key("id", AttributeValue::S("warmup".into())) + .send() + .await + .unwrap(); + sdk.batch_write_item() + .set_request_items(Some(HashMap::from([( + "FollowerRecovery".into(), + ["batch-a", "batch-b"] + .map(|id| { + WriteRequest::builder() + .put_request( + PutRequest::builder() + .item("id", AttributeValue::S(id.into())) + .build() + .unwrap(), + ) + .build() + }) + .to_vec(), + )]))) + .send() + .await + .unwrap(); + let transaction = ["tx-a", "tx-b"] + .map(|id| { + TransactWriteItem::builder() + .put( + aws_sdk_dynamodb::types::Put::builder() + .table_name("FollowerRecovery") + .item("id", AttributeValue::S(id.into())) + .condition_expression("attribute_not_exists(id)") + .build() + .unwrap(), + ) + .build() + }) + .to_vec(); + sdk.transact_write_items() + .client_request_token("follower-process-replay") + .set_transact_items(Some(transaction.clone())) + .send() + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(5), barrier.blocked.cancelled()) + .await + .unwrap(); + assert_eq!( + authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .root + .as_ref() + .unwrap() + .commit_sequence, + predecessor + ); + let failed = followers + .iter() + .position(|(_, _, _, peer, _)| owner.endpoint == format!("https://{peer}")); + let (child, path, log, public) = match failed { + Some(index) => { + let (child, path, log, _, public) = &mut followers[index]; + (child, path.as_path(), log.as_path(), *public) + } + None => { + assert_eq!(owner.endpoint, format!("https://{}", fixture.peer)); + ( + &mut fixture.child, + fixture.config.as_path(), + fixture.log.as_path(), + fixture.public, + ) + } + }; + child.kill().unwrap(); + child.wait().unwrap(); + assert_eq!( + authority + .load(target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .root + .as_ref() + .unwrap() + .commit_sequence, + predecessor + ); + barrier.release.cancel(); + tokio::time::timeout(Duration::from_secs(20), async { + while directory.is_live(owner.session, now_ms()).await.unwrap() { + tokio::time::sleep(Duration::from_millis(100)).await; + } + }) + .await + .unwrap(); + // Recover all Cells acknowledged by the failed boot, even if placement + // originally chose a follower node as this table's owner. + let mut successor: serde_json::Value = + serde_json::from_slice(&fs::read(path).unwrap()).unwrap(); + successor["owned_accounts"] = config["owned_accounts"].clone(); + fs::write(path, successor.to_string()).unwrap(); + *child = start(path, log, false, proxy_address); + wait_healthy(child, public, log); + let read = sdk + .get_item() + .table_name("FollowerRecovery") + .key("id", item["id"].clone()) + .consistent_read(true) + .send() + .await + .unwrap(); + assert_eq!(read.item(), Some(&item)); + let restored = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_ne!( + restored.value().owner.as_ref().unwrap().session, + owner.session + ); + // Replaying the saved transaction outcome must not re-evaluate its insert + // conditions or emit another stream record. + sdk.transact_write_items() + .client_request_token("follower-process-replay") + .set_transact_items(Some(transaction)) + .send() + .await + .unwrap(); + assert!( + sdk.get_item() + .table_name("FollowerRecovery") + .key("id", AttributeValue::S("warmup".into())) + .consistent_read(true) + .send() + .await + .unwrap() + .item() + .is_none() + ); + for id in ["batch-a", "batch-b", "tx-a", "tx-b"] { + assert_eq!( + sdk.get_item() + .table_name("FollowerRecovery") + .key("id", AttributeValue::S(id.into())) + .consistent_read(true) + .send() + .await + .unwrap() + .item() + .unwrap() + .get("id"), + Some(&AttributeValue::S(id.into())) + ); + } + let described = streams_cli(fixture.public, &["describe-stream", "--stream-arn", arn]); + let mut records = Vec::new(); + for shard in described["StreamDescription"]["Shards"].as_array().unwrap() { + let iterator = streams_cli( + fixture.public, + &[ + "get-shard-iterator", + "--stream-arn", + arn, + "--shard-id", + shard["ShardId"].as_str().unwrap(), + "--shard-iterator-type", + "TRIM_HORIZON", + ], + ); + records.extend( + streams_cli( + fixture.public, + &[ + "get-records", + "--shard-iterator", + iterator["ShardIterator"].as_str().unwrap(), + ], + )["Records"] + .as_array() + .unwrap() + .clone(), + ); + } + for (id, event) in [ + ("acknowledged", "INSERT"), + ("acknowledged", "MODIFY"), + ("warmup", "REMOVE"), + ("batch-a", "INSERT"), + ("batch-b", "INSERT"), + ("tx-a", "INSERT"), + ("tx-b", "INSERT"), + ] { + assert_eq!( + records + .iter() + .filter(|record| record["dynamodb"]["Keys"]["id"]["S"] == id + && record["eventName"] == event) + .count(), + 1, + "missing or duplicate {event} for {id}" + ); + } + for (child, _, log, _, _) in &mut followers { + stop(child, log); + } + stop(&mut fixture.child, &fixture.log); + proxy_server.abort(); +} diff --git a/tests/server_binary/large_reads.rs b/tests/server_binary/large_reads.rs index 9d50322..d3d3b95 100644 --- a/tests/server_binary/large_reads.rs +++ b/tests/server_binary/large_reads.rs @@ -7,7 +7,7 @@ use aws_sdk_dynamodb::types::{ #[tokio::test(flavor = "multi_thread", worker_threads = 2)] #[ignore = "requires Docker, aws CLI, and the pinned RustFS GA image"] -async fn oversized_single_cell_transaction_read_uses_saved_images() { +async fn large_single_cell_transaction_read_survives_owner_restart() { let mut fixture = crate::process_fixture_with_cache(1, true).await; let table = "ProcessLargeRead"; fixture @@ -67,13 +67,28 @@ async fn oversized_single_cell_transaction_read_uses_saved_images() { let result = fixture .sdk .transact_get_items() - .set_transact_items(Some(reads)) + .set_transact_items(Some(reads.clone())) .send() .await .unwrap(); assert_eq!(result.responses().len(), expected.len()); - for (response, item) in result.responses().iter().zip(expected) { - assert_eq!(response.item(), Some(&item)); + for (response, item) in result.responses().iter().zip(&expected) { + assert_eq!(response.item(), Some(item)); + } + fixture.child.kill().unwrap(); + fixture.child.wait().unwrap(); + fixture.child = crate::start(&fixture.config, &fixture.log, false, fixture.s3); + crate::wait_healthy(&mut fixture.child, fixture.public, &fixture.log); + let restored = fixture + .sdk + .transact_get_items() + .set_transact_items(Some(reads)) + .send() + .await + .unwrap(); + for (response, item) in restored.responses().iter().zip(&expected) { + assert_eq!(response.item(), Some(item)); } + assert_eq!(restored.responses().len(), expected.len()); crate::stop(&mut fixture.child, &fixture.log); } diff --git a/tests/server_binary/metadata_cache.rs b/tests/server_binary/metadata_cache.rs index c003ec6..7bc8b75 100644 --- a/tests/server_binary/metadata_cache.rs +++ b/tests/server_binary/metadata_cache.rs @@ -71,5 +71,141 @@ async fn cached_metadata_observes_local_create_and_update() { .and_then(|table| table.deletion_protection_enabled()), Some(true), ); + + // BatchGetItem reaches the storage table-info lookup directly. Warm its + // backend cache, then ensure local recreation replaces that generation. + for id in ["first", "second"] { + fixture + .sdk + .put_item() + .table_name("ProcessData") + .item("pk", AttributeValue::S(id.into())) + .item("value", AttributeValue::S("old".into())) + .send() + .await + .unwrap(); + } + let batch = || { + HashMap::from([( + "ProcessData".into(), + KeysAndAttributes::builder() + .keys(HashMap::from([( + "pk".into(), + AttributeValue::S("first".into()), + )])) + .keys(HashMap::from([( + "pk".into(), + AttributeValue::S("second".into()), + )])) + .consistent_read(true) + .build() + .unwrap(), + )]) + }; + let old = fixture + .sdk + .batch_get_item() + .set_request_items(Some(batch())) + .send() + .await + .unwrap(); + assert_eq!(old.responses().unwrap()["ProcessData"].len(), 2); + assert!( + old.responses().unwrap()["ProcessData"] + .iter() + .all(|item| item.get("value") == Some(&AttributeValue::S("old".into()))) + ); + fixture + .sdk + .update_table() + .table_name("ProcessData") + .deletion_protection_enabled(false) + .send() + .await + .unwrap(); + fixture + .sdk + .delete_table() + .table_name("ProcessData") + .send() + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(45), async { + loop { + match fixture + .sdk + .describe_table() + .table_name("ProcessData") + .send() + .await + { + Err(error) + if error + .as_service_error() + .is_some_and(|error| error.is_resource_not_found_exception()) => + { + break; + } + Ok(_) => tokio::time::sleep(Duration::from_millis(50)).await, + Err(error) => panic!("delete did not settle: {error:?}"), + } + } + }) + .await + .unwrap(); + let recreated = fixture + .sdk + .create_table() + .table_name("ProcessData") + .key_schema( + KeySchemaElement::builder() + .attribute_name("pk") + .key_type(KeyType::Hash) + .build() + .unwrap(), + ) + .attribute_definitions( + AttributeDefinition::builder() + .attribute_name("pk") + .attribute_type(ScalarAttributeType::S) + .build() + .unwrap(), + ) + .billing_mode(BillingMode::PayPerRequest) + .send() + .await + .unwrap(); + assert_ne!( + recreated.table_description().unwrap().table_id(), + before_update.table().unwrap().table_id() + ); + for id in ["first", "second"] { + fixture + .sdk + .put_item() + .table_name("ProcessData") + .item("pk", AttributeValue::S(id.into())) + .item("value", AttributeValue::S("new".into())) + .send() + .await + .unwrap(); + } + // The published metadata and item values must survive a real owner restart. + stop(&mut fixture.child, &fixture.log); + fixture.child = start(&fixture.config, &fixture.log, false, fixture.s3); + wait_healthy(&mut fixture.child, fixture.public, &fixture.log); + let new = fixture + .sdk + .batch_get_item() + .set_request_items(Some(batch())) + .send() + .await + .unwrap(); + assert_eq!(new.responses().unwrap()["ProcessData"].len(), 2); + assert!( + new.responses().unwrap()["ProcessData"] + .iter() + .all(|item| item.get("value") == Some(&AttributeValue::S("new".into()))) + ); stop(&mut fixture.child, &fixture.log); }